Files
openworker/coworker/permissions.py
T
Devika Verma a442e0e0b8 PR1: split egress out of READ, stop override downgrades, scope patch writes
Three gate defects, each verified by direct execution before and after.

1. web_fetch was RiskClass.READ, so is_consequential() was False and evaluate()
   returned allow on its third rung -- before any rule, mode or PDP, in EVERY
   mode including plan/discuss. A URL's query string carries data outbound, so
   this was an ungated egress path. New RiskClass.EGRESS covers model-chosen
   network reads; web_search stays READ (fixed configured provider, not a
   model-chosen host). Adds an allowed_domains allowlist (exact host or
   subdomain; 'evil-python.org' never matches 'python.org'), a session-scoped
   "always allow this domain" grant, and ApprovalOutcome.ALWAYS_DOMAIN.

2. A risk override could DOWNGRADE a built-in: marking write_file as read made
   is_write False (skipping path scoping) and consequential False (skipping the
   read-only gate) at once -- one settings line disabling two protections, in
   every future session. Overrides may now only tighten a built-in write/exec/
   egress tool; relaxing a metadata/MCP tool (the intended use) still works.

3. Path scoping read a literal "path" argument, so apply_patch and
   apply_unified_diff -- whose paths live inside the patch/diff blob -- were
   never scoped at all. write_paths() extracts them from the blob and scopes
   every one; a write whose path cannot be located now fails closed to approval
   rather than slipping through auto/custom unscoped.

allowed_domains is user-global only, alongside auto_allow: a cloned repo must
not be able to widen the agent's network reach.

Golden matrix: web_fetch interactive allow->ask, plan allow->deny, plus new
egress/patch rows (31 rows green). test_permissions_risk's override test
asserted the old downgrade behavior and is updated to the tightening rule.
Full suite: 22 failures, all pre-existing on the unmodified tree (boto3 absent,
Windows symlink privilege, Slack socket timeouts) -- none introduced here.

Design of record: ocw-context/docs/reviewed-auto-mode.md Part 3.
2026-08-11 11:37:46 -07:00

329 lines
14 KiB
Python

"""Permission engine — decides allow / deny / ask-user for each proposed tool call.
Modes: Plan (read-only) · Interactive (auto reads, ask on writes/commands) · Auto
(allow, still path-scoped). Refined by argument patterns (path-under-root, command
prefixes) and a session allowlist. The engine only *decides*; the turn engine routes
`needs_user` decisions to a surface for approval and records the outcome.
"""
from __future__ import annotations
import re
import shlex
from dataclasses import dataclass, field
from enum import Enum
from pathlib import Path
from typing import Any, Optional
from urllib.parse import urlsplit
# Shell metacharacters that turn one "allowlisted" command into several. Any of these in a
# command disqualifies it from allowlist auto-run — approval is required instead. Covers
# chaining (`;` `&` `&&` `||`), pipes (`|`), redirection (`>` `<`), command substitution
# (`` ` `` `$(`), process substitution / grouping (`(`), and newlines.
_SHELL_OPERATORS = (";", "&", "|", ">", "<", "`", "$(", "(", "\n", "\r")
def _has_shell_operators(command: str) -> bool:
return any(op in command for op in _SHELL_OPERATORS)
def _host_of(url_or_domain: str) -> str:
"""The lowercased host of a URL, or a bare domain as-is. `''` when there's nothing
usable. Accepts both `https://docs.python.org/x` and `docs.python.org`."""
s = (url_or_domain or "").strip().lower()
if not s:
return ""
if "://" in s:
return urlsplit(s).hostname or ""
return urlsplit("//" + s).hostname or s
# The argument that names a write tool's target path, when it's a single top-level field.
# Patch/diff tools carry their paths inside the blob instead — extracted in `write_paths`.
_PATH_ARG: dict[str, str] = {"write_file": "path", "replace_in_file": "path"}
# apply_patch (Codex format) file headers, and unified-diff `+++ b/<path>` headers.
_APPLY_PATCH_FILE = re.compile(
r"^\*\*\* (?:Add|Update|Delete) File: (.+)$", re.MULTILINE
)
_APPLY_PATCH_MOVE = re.compile(r"^\*\*\* Move to: (.+)$", re.MULTILINE)
_UNIFIED_DIFF_FILE = re.compile(r"^\+\+\+ (?:b/)?(.+?)\s*$", re.MULTILINE)
def write_paths(tool_name: str, arguments: dict[str, Any]) -> tuple[list[str], bool]:
"""Every filesystem path a write tool would touch, for root scoping.
Returns ``(paths, located)``. ``located`` is False when the path can't be determined
(an unknown write tool, or a patch/diff blob with no parseable file header) — the caller
must then fail closed rather than skip scoping, so an unscoped write can't slip through
auto/custom mode.
"""
arg = _PATH_ARG.get(tool_name)
if arg is not None:
value = arguments.get(arg)
return ([str(value)], True) if value else ([], False)
if tool_name == "apply_patch":
blob = str(arguments.get("patch", ""))
paths = _APPLY_PATCH_FILE.findall(blob) + _APPLY_PATCH_MOVE.findall(blob)
return ([p.strip() for p in paths], bool(paths))
if tool_name == "apply_unified_diff":
blob = str(arguments.get("diff", ""))
paths = [p for p in _UNIFIED_DIFF_FILE.findall(blob) if p and p != "/dev/null"]
return (paths, bool(paths))
# Unknown write tool (e.g. one promoted to write via a user override): we cannot locate
# its path, so it cannot be auto-scoped.
return ([], False)
from .risk import ( # re-exported for back-compat (manager.py imports WRITE_TOOLS)
SHELL_TOOL,
WRITE_TOOLS,
RiskClass,
RiskOverrides,
classify,
is_consequential,
)
class Mode(str, Enum):
DISCUSS = "discuss" # read-only conversation: no edits, no planning workflow
PLAN = (
"plan" # read-only + the planning contract (explore → propose_plan → execute)
)
INTERACTIVE = "interactive" # ask for approval (default)
AUTO = "auto" # full access
CUSTOM = "custom" # interactive + auto-allow the config's `auto_allow` tools
# Modes whose enforcement is read-only. DISCUSS and PLAN share the same gate; they differ
# only in intent — PLAN additionally drives the agent toward a propose_plan approval.
READ_ONLY_MODES = frozenset({Mode.DISCUSS, Mode.PLAN})
@dataclass
class Decision:
allowed: bool
reason: str = ""
needs_user: bool = False # True → surface should prompt the user for approval
# Set when a task-scoped standing rule allowed the call ("tool → target") so the
# engine can audit the exact rule and the tool card can say so (§25).
rule: str = ""
def standing_rule_candidate(
tool_name: str,
arguments: dict[str, Any],
metadata: Any = None,
overrides: Optional[RiskOverrides] = None,
) -> Optional[str]:
"""The target value iff this call is eligible for a task-scoped standing rule
(UX-DECISIONS §25): external-risk only (never exec/write-local — shell asks forever),
the tool must declare a target argument, and the call must actually name a target.
Returns None otherwise — ineligible calls keep parking approvals as today."""
from .connectors.tool_defs import target_arg_for
if classify(tool_name, metadata, overrides) is not RiskClass.EXTERNAL:
return None
arg = target_arg_for(tool_name)
if arg is None:
return None
value = str((arguments or {}).get(arg) or "").strip()
return value or None
@dataclass
class PermissionEngine:
workspace_root: Path
mode: Mode = Mode.INTERACTIVE
allowed_commands: list[str] = field(default_factory=list)
auto_allow_tools: set[str] = field(default_factory=set)
session_allow_tools: set[str] = field(default_factory=set)
session_allow_commands: set[str] = field(default_factory=set)
# Egress domains that auto-run without a prompt: `allowed_domains` from user config, plus
# `session_allow_domains` minted by "Always allow this domain". Matched by exact host or
# subdomain suffix (see `_domain_allowed`).
allowed_domains: list[str] = field(default_factory=list)
session_allow_domains: set[str] = field(default_factory=set)
# Task-scoped standing rules (§25): {tool: {allowed targets}}, seeded from the owning
# ScheduledTask's target-shaped entries. Kept by reference and re-read every check, so a
# rule minted mid-run ("Allow every time") applies to the run's next call too.
task_rules: dict[str, set[str]] = field(default_factory=dict)
# User-local risk override resolver (Phase 2). None → use the base classification.
risk_overrides: Optional[RiskOverrides] = None
# Shared, possibly-mutable list of roots (RootDir-like / dicts). When omitted, the single
# `workspace_root` is the sole writable root (back-compat). Kept by reference and re-read on
# every check, so runtime add/remove of folders takes effect without rebuilding the engine.
roots: Optional[list] = None
def __post_init__(self) -> None:
self.workspace_root = Path(self.workspace_root).expanduser().resolve()
self.auto_allow_tools = set(self.auto_allow_tools)
if self.roots is None:
self.roots = [{"path": self.workspace_root, "writable": True}]
def _resolved_roots(self) -> list[tuple[Path, bool]]:
out: list[tuple[Path, bool]] = []
for r in self.roots or []:
if isinstance(r, dict):
p, w = r["path"], bool(r.get("writable", False))
elif isinstance(r, (str, Path)):
p, w = r, True
else: # duck-typed RootDir-like
p, w = getattr(r, "path"), bool(getattr(r, "writable", False))
out.append((Path(p).expanduser().resolve(), w))
return out
def evaluate(
self, tool_name: str, arguments: dict[str, Any], metadata: Any = None
) -> Decision:
arguments = arguments or {}
is_connector = getattr(metadata, "category", "") == "connector"
risk = classify(tool_name, metadata, self.risk_overrides)
is_write = risk is RiskClass.WRITE_LOCAL
is_shell = risk is RiskClass.EXEC
is_egress = risk is RiskClass.EGRESS
consequential = is_consequential(risk)
# Discuss / plan modes: read-only.
if self.mode in READ_ONLY_MODES and consequential:
return Decision(
False, f"{self.mode.value} mode is read-only", needs_user=False
)
# Path scoping for writes (all modes): every path the write touches must land in a
# writable root. A write whose path can't be located is not scoped-able, so it fails
# closed to approval rather than slipping through auto/custom unscoped.
if is_write:
paths, located = write_paths(tool_name, arguments)
if not located:
return Decision(
False,
"cannot determine the write path to scope",
needs_user=True,
)
for path in paths:
if not self._under_writable_root(path):
return Decision(
False, f"path is not in a writable directory: {path}"
)
# Non-consequential tools always run.
if not consequential:
return Decision(True, "low risk")
# Full access.
if self.mode is Mode.AUTO:
return Decision(True, "full access")
# interactive / custom: allowlists.
if is_shell:
command = str(arguments.get("command", ""))
if self._command_allowed(command):
return Decision(True, "command on allowlist")
if command and command in self.session_allow_commands:
return Decision(True, "command allowed for session")
if is_egress:
url = str(arguments.get("url", ""))
if self._domain_allowed(url):
return Decision(True, "domain on allowlist")
if tool_name in self.session_allow_tools and not is_connector:
return Decision(True, "tool allowed for session")
# Task-scoped standing rules (§25): tool + exact target, owned by the automation.
# Deliberately NOT subject to the connector exclusion above — the exact-target
# binding is what makes auto-allowing a connector tool safe. Never for exec risk
# (candidate extraction is external-risk-only), and additive on top of the mode:
# read-only modes already returned before this point.
if tool_name in self.task_rules:
target = standing_rule_candidate(
tool_name, arguments, metadata, self.risk_overrides
)
if target and target in self.task_rules[tool_name]:
rule = f"{tool_name}{target}"
return Decision(True, f"allowed by standing rule: {rule}", rule=rule)
# Custom mode auto-approves the configured tools.
if self.mode is Mode.CUSTOM and tool_name in self.auto_allow_tools:
return Decision(True, "auto-allowed by config")
# Otherwise: ask the user.
return Decision(False, "requires approval", needs_user=True)
# -- session memory ---------------------------------------------------------
def allow_tool_for_session(self, tool_name: str) -> None:
self.session_allow_tools.add(tool_name)
def allow_command_for_session(self, command: str) -> None:
if command:
self.session_allow_commands.add(command)
def allow_domain_for_session(self, url_or_domain: str) -> None:
"""Remember an egress destination for this session ("Always allow this domain")."""
host = _host_of(url_or_domain)
if host:
self.session_allow_domains.add(host)
# -- helpers ----------------------------------------------------------------
def _candidate(self, path: str) -> Path:
# Relative paths resolve against the primary (workspace_root); absolute/`~` taken as-is.
p = Path(path).expanduser()
return p.resolve() if p.is_absolute() else (self.workspace_root / p).resolve()
def _under_root(self, path: str) -> bool:
candidate = self._candidate(path)
for rp, _ in self._resolved_roots():
try:
candidate.relative_to(rp)
return True
except ValueError:
continue
return False
def _under_writable_root(self, path: str) -> bool:
candidate = self._candidate(path)
for rp, writable in self._resolved_roots():
if not writable:
continue
try:
candidate.relative_to(rp)
return True
except ValueError:
continue
return False
def _domain_allowed(self, url: str) -> bool:
"""True when the URL's host is an allowed egress destination — an exact match or a
subdomain of an allowed domain (so `docs.python.org` matches `python.org`, but
`evil-python.org` never matches `python.org`)."""
host = _host_of(url)
if not host:
return False
allowed = {d for d in (_host_of(x) for x in self.allowed_domains) if d}
allowed |= self.session_allow_domains
for dom in allowed:
if host == dom or host.endswith("." + dom):
return True
return False
def _command_allowed(self, command: str) -> bool:
# An allowlist entry auto-runs a command WITHOUT approval, so prefix matching is
# unsafe: `git status` would auto-approve `git status && rm -rf ~`. Reject anything
# carrying shell operators (chaining/redirection/substitution) up front, then match
# the parsed argv against each entry — the entry's own tokens must be an exact
# prefix of the command's tokens (so `git status` matches `git status -s` but never
# `git statusfoo` or a bare `git`).
if _has_shell_operators(command):
return False
try:
argv = shlex.split(command)
except ValueError:
return False # unbalanced quotes etc. — treat as not-allowlisted
if not argv:
return False
for allowed in self.allowed_commands:
try:
prefix = shlex.split(allowed)
except ValueError:
continue
if prefix and argv[: len(prefix)] == prefix:
return True
return False