mirror of
https://github.com/andrewyng/openworker.git
synced 2026-09-13 15:50:02 +00:00
Three gate defects, each verified by direct execution before and after. 1. web_fetch was RiskClass.READ, so is_consequential() was False and evaluate() returned allow on its third rung -- before any rule, mode or PDP, in EVERY mode including plan/discuss. A URL's query string carries data outbound, so this was an ungated egress path. New RiskClass.EGRESS covers model-chosen network reads; web_search stays READ (fixed configured provider, not a model-chosen host). Adds an allowed_domains allowlist (exact host or subdomain; 'evil-python.org' never matches 'python.org'), a session-scoped "always allow this domain" grant, and ApprovalOutcome.ALWAYS_DOMAIN. 2. A risk override could DOWNGRADE a built-in: marking write_file as read made is_write False (skipping path scoping) and consequential False (skipping the read-only gate) at once -- one settings line disabling two protections, in every future session. Overrides may now only tighten a built-in write/exec/ egress tool; relaxing a metadata/MCP tool (the intended use) still works. 3. Path scoping read a literal "path" argument, so apply_patch and apply_unified_diff -- whose paths live inside the patch/diff blob -- were never scoped at all. write_paths() extracts them from the blob and scopes every one; a write whose path cannot be located now fails closed to approval rather than slipping through auto/custom unscoped. allowed_domains is user-global only, alongside auto_allow: a cloned repo must not be able to widen the agent's network reach. Golden matrix: web_fetch interactive allow->ask, plan allow->deny, plus new egress/patch rows (31 rows green). test_permissions_risk's override test asserted the old downgrade behavior and is updated to the tightening rule. Full suite: 22 failures, all pre-existing on the unmodified tree (boto3 absent, Windows symlink privilege, Slack socket timeouts) -- none introduced here. Design of record: ocw-context/docs/reviewed-auto-mode.md Part 3.
329 lines
14 KiB
Python
329 lines
14 KiB
Python
"""Permission engine — decides allow / deny / ask-user for each proposed tool call.
|
|
|
|
Modes: Plan (read-only) · Interactive (auto reads, ask on writes/commands) · Auto
|
|
(allow, still path-scoped). Refined by argument patterns (path-under-root, command
|
|
prefixes) and a session allowlist. The engine only *decides*; the turn engine routes
|
|
`needs_user` decisions to a surface for approval and records the outcome.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import re
|
|
import shlex
|
|
from dataclasses import dataclass, field
|
|
from enum import Enum
|
|
from pathlib import Path
|
|
from typing import Any, Optional
|
|
from urllib.parse import urlsplit
|
|
|
|
# Shell metacharacters that turn one "allowlisted" command into several. Any of these in a
|
|
# command disqualifies it from allowlist auto-run — approval is required instead. Covers
|
|
# chaining (`;` `&` `&&` `||`), pipes (`|`), redirection (`>` `<`), command substitution
|
|
# (`` ` `` `$(`), process substitution / grouping (`(`), and newlines.
|
|
_SHELL_OPERATORS = (";", "&", "|", ">", "<", "`", "$(", "(", "\n", "\r")
|
|
|
|
|
|
def _has_shell_operators(command: str) -> bool:
|
|
return any(op in command for op in _SHELL_OPERATORS)
|
|
|
|
|
|
def _host_of(url_or_domain: str) -> str:
|
|
"""The lowercased host of a URL, or a bare domain as-is. `''` when there's nothing
|
|
usable. Accepts both `https://docs.python.org/x` and `docs.python.org`."""
|
|
s = (url_or_domain or "").strip().lower()
|
|
if not s:
|
|
return ""
|
|
if "://" in s:
|
|
return urlsplit(s).hostname or ""
|
|
return urlsplit("//" + s).hostname or s
|
|
|
|
|
|
# The argument that names a write tool's target path, when it's a single top-level field.
|
|
# Patch/diff tools carry their paths inside the blob instead — extracted in `write_paths`.
|
|
_PATH_ARG: dict[str, str] = {"write_file": "path", "replace_in_file": "path"}
|
|
# apply_patch (Codex format) file headers, and unified-diff `+++ b/<path>` headers.
|
|
_APPLY_PATCH_FILE = re.compile(
|
|
r"^\*\*\* (?:Add|Update|Delete) File: (.+)$", re.MULTILINE
|
|
)
|
|
_APPLY_PATCH_MOVE = re.compile(r"^\*\*\* Move to: (.+)$", re.MULTILINE)
|
|
_UNIFIED_DIFF_FILE = re.compile(r"^\+\+\+ (?:b/)?(.+?)\s*$", re.MULTILINE)
|
|
|
|
|
|
def write_paths(tool_name: str, arguments: dict[str, Any]) -> tuple[list[str], bool]:
|
|
"""Every filesystem path a write tool would touch, for root scoping.
|
|
|
|
Returns ``(paths, located)``. ``located`` is False when the path can't be determined
|
|
(an unknown write tool, or a patch/diff blob with no parseable file header) — the caller
|
|
must then fail closed rather than skip scoping, so an unscoped write can't slip through
|
|
auto/custom mode.
|
|
"""
|
|
arg = _PATH_ARG.get(tool_name)
|
|
if arg is not None:
|
|
value = arguments.get(arg)
|
|
return ([str(value)], True) if value else ([], False)
|
|
if tool_name == "apply_patch":
|
|
blob = str(arguments.get("patch", ""))
|
|
paths = _APPLY_PATCH_FILE.findall(blob) + _APPLY_PATCH_MOVE.findall(blob)
|
|
return ([p.strip() for p in paths], bool(paths))
|
|
if tool_name == "apply_unified_diff":
|
|
blob = str(arguments.get("diff", ""))
|
|
paths = [p for p in _UNIFIED_DIFF_FILE.findall(blob) if p and p != "/dev/null"]
|
|
return (paths, bool(paths))
|
|
# Unknown write tool (e.g. one promoted to write via a user override): we cannot locate
|
|
# its path, so it cannot be auto-scoped.
|
|
return ([], False)
|
|
|
|
from .risk import ( # re-exported for back-compat (manager.py imports WRITE_TOOLS)
|
|
SHELL_TOOL,
|
|
WRITE_TOOLS,
|
|
RiskClass,
|
|
RiskOverrides,
|
|
classify,
|
|
is_consequential,
|
|
)
|
|
|
|
|
|
class Mode(str, Enum):
|
|
DISCUSS = "discuss" # read-only conversation: no edits, no planning workflow
|
|
PLAN = (
|
|
"plan" # read-only + the planning contract (explore → propose_plan → execute)
|
|
)
|
|
INTERACTIVE = "interactive" # ask for approval (default)
|
|
AUTO = "auto" # full access
|
|
CUSTOM = "custom" # interactive + auto-allow the config's `auto_allow` tools
|
|
|
|
|
|
# Modes whose enforcement is read-only. DISCUSS and PLAN share the same gate; they differ
|
|
# only in intent — PLAN additionally drives the agent toward a propose_plan approval.
|
|
READ_ONLY_MODES = frozenset({Mode.DISCUSS, Mode.PLAN})
|
|
|
|
|
|
@dataclass
|
|
class Decision:
|
|
allowed: bool
|
|
reason: str = ""
|
|
needs_user: bool = False # True → surface should prompt the user for approval
|
|
# Set when a task-scoped standing rule allowed the call ("tool → target") so the
|
|
# engine can audit the exact rule and the tool card can say so (§25).
|
|
rule: str = ""
|
|
|
|
|
|
def standing_rule_candidate(
|
|
tool_name: str,
|
|
arguments: dict[str, Any],
|
|
metadata: Any = None,
|
|
overrides: Optional[RiskOverrides] = None,
|
|
) -> Optional[str]:
|
|
"""The target value iff this call is eligible for a task-scoped standing rule
|
|
(UX-DECISIONS §25): external-risk only (never exec/write-local — shell asks forever),
|
|
the tool must declare a target argument, and the call must actually name a target.
|
|
Returns None otherwise — ineligible calls keep parking approvals as today."""
|
|
from .connectors.tool_defs import target_arg_for
|
|
|
|
if classify(tool_name, metadata, overrides) is not RiskClass.EXTERNAL:
|
|
return None
|
|
arg = target_arg_for(tool_name)
|
|
if arg is None:
|
|
return None
|
|
value = str((arguments or {}).get(arg) or "").strip()
|
|
return value or None
|
|
|
|
|
|
@dataclass
|
|
class PermissionEngine:
|
|
workspace_root: Path
|
|
mode: Mode = Mode.INTERACTIVE
|
|
allowed_commands: list[str] = field(default_factory=list)
|
|
auto_allow_tools: set[str] = field(default_factory=set)
|
|
session_allow_tools: set[str] = field(default_factory=set)
|
|
session_allow_commands: set[str] = field(default_factory=set)
|
|
# Egress domains that auto-run without a prompt: `allowed_domains` from user config, plus
|
|
# `session_allow_domains` minted by "Always allow this domain". Matched by exact host or
|
|
# subdomain suffix (see `_domain_allowed`).
|
|
allowed_domains: list[str] = field(default_factory=list)
|
|
session_allow_domains: set[str] = field(default_factory=set)
|
|
# Task-scoped standing rules (§25): {tool: {allowed targets}}, seeded from the owning
|
|
# ScheduledTask's target-shaped entries. Kept by reference and re-read every check, so a
|
|
# rule minted mid-run ("Allow every time") applies to the run's next call too.
|
|
task_rules: dict[str, set[str]] = field(default_factory=dict)
|
|
# User-local risk override resolver (Phase 2). None → use the base classification.
|
|
risk_overrides: Optional[RiskOverrides] = None
|
|
# Shared, possibly-mutable list of roots (RootDir-like / dicts). When omitted, the single
|
|
# `workspace_root` is the sole writable root (back-compat). Kept by reference and re-read on
|
|
# every check, so runtime add/remove of folders takes effect without rebuilding the engine.
|
|
roots: Optional[list] = None
|
|
|
|
def __post_init__(self) -> None:
|
|
self.workspace_root = Path(self.workspace_root).expanduser().resolve()
|
|
self.auto_allow_tools = set(self.auto_allow_tools)
|
|
if self.roots is None:
|
|
self.roots = [{"path": self.workspace_root, "writable": True}]
|
|
|
|
def _resolved_roots(self) -> list[tuple[Path, bool]]:
|
|
out: list[tuple[Path, bool]] = []
|
|
for r in self.roots or []:
|
|
if isinstance(r, dict):
|
|
p, w = r["path"], bool(r.get("writable", False))
|
|
elif isinstance(r, (str, Path)):
|
|
p, w = r, True
|
|
else: # duck-typed RootDir-like
|
|
p, w = getattr(r, "path"), bool(getattr(r, "writable", False))
|
|
out.append((Path(p).expanduser().resolve(), w))
|
|
return out
|
|
|
|
def evaluate(
|
|
self, tool_name: str, arguments: dict[str, Any], metadata: Any = None
|
|
) -> Decision:
|
|
arguments = arguments or {}
|
|
is_connector = getattr(metadata, "category", "") == "connector"
|
|
risk = classify(tool_name, metadata, self.risk_overrides)
|
|
is_write = risk is RiskClass.WRITE_LOCAL
|
|
is_shell = risk is RiskClass.EXEC
|
|
is_egress = risk is RiskClass.EGRESS
|
|
consequential = is_consequential(risk)
|
|
|
|
# Discuss / plan modes: read-only.
|
|
if self.mode in READ_ONLY_MODES and consequential:
|
|
return Decision(
|
|
False, f"{self.mode.value} mode is read-only", needs_user=False
|
|
)
|
|
|
|
# Path scoping for writes (all modes): every path the write touches must land in a
|
|
# writable root. A write whose path can't be located is not scoped-able, so it fails
|
|
# closed to approval rather than slipping through auto/custom unscoped.
|
|
if is_write:
|
|
paths, located = write_paths(tool_name, arguments)
|
|
if not located:
|
|
return Decision(
|
|
False,
|
|
"cannot determine the write path to scope",
|
|
needs_user=True,
|
|
)
|
|
for path in paths:
|
|
if not self._under_writable_root(path):
|
|
return Decision(
|
|
False, f"path is not in a writable directory: {path}"
|
|
)
|
|
|
|
# Non-consequential tools always run.
|
|
if not consequential:
|
|
return Decision(True, "low risk")
|
|
|
|
# Full access.
|
|
if self.mode is Mode.AUTO:
|
|
return Decision(True, "full access")
|
|
|
|
# interactive / custom: allowlists.
|
|
if is_shell:
|
|
command = str(arguments.get("command", ""))
|
|
if self._command_allowed(command):
|
|
return Decision(True, "command on allowlist")
|
|
if command and command in self.session_allow_commands:
|
|
return Decision(True, "command allowed for session")
|
|
if is_egress:
|
|
url = str(arguments.get("url", ""))
|
|
if self._domain_allowed(url):
|
|
return Decision(True, "domain on allowlist")
|
|
if tool_name in self.session_allow_tools and not is_connector:
|
|
return Decision(True, "tool allowed for session")
|
|
|
|
# Task-scoped standing rules (§25): tool + exact target, owned by the automation.
|
|
# Deliberately NOT subject to the connector exclusion above — the exact-target
|
|
# binding is what makes auto-allowing a connector tool safe. Never for exec risk
|
|
# (candidate extraction is external-risk-only), and additive on top of the mode:
|
|
# read-only modes already returned before this point.
|
|
if tool_name in self.task_rules:
|
|
target = standing_rule_candidate(
|
|
tool_name, arguments, metadata, self.risk_overrides
|
|
)
|
|
if target and target in self.task_rules[tool_name]:
|
|
rule = f"{tool_name} → {target}"
|
|
return Decision(True, f"allowed by standing rule: {rule}", rule=rule)
|
|
|
|
# Custom mode auto-approves the configured tools.
|
|
if self.mode is Mode.CUSTOM and tool_name in self.auto_allow_tools:
|
|
return Decision(True, "auto-allowed by config")
|
|
|
|
# Otherwise: ask the user.
|
|
return Decision(False, "requires approval", needs_user=True)
|
|
|
|
# -- session memory ---------------------------------------------------------
|
|
def allow_tool_for_session(self, tool_name: str) -> None:
|
|
self.session_allow_tools.add(tool_name)
|
|
|
|
def allow_command_for_session(self, command: str) -> None:
|
|
if command:
|
|
self.session_allow_commands.add(command)
|
|
|
|
def allow_domain_for_session(self, url_or_domain: str) -> None:
|
|
"""Remember an egress destination for this session ("Always allow this domain")."""
|
|
host = _host_of(url_or_domain)
|
|
if host:
|
|
self.session_allow_domains.add(host)
|
|
|
|
# -- helpers ----------------------------------------------------------------
|
|
def _candidate(self, path: str) -> Path:
|
|
# Relative paths resolve against the primary (workspace_root); absolute/`~` taken as-is.
|
|
p = Path(path).expanduser()
|
|
return p.resolve() if p.is_absolute() else (self.workspace_root / p).resolve()
|
|
|
|
def _under_root(self, path: str) -> bool:
|
|
candidate = self._candidate(path)
|
|
for rp, _ in self._resolved_roots():
|
|
try:
|
|
candidate.relative_to(rp)
|
|
return True
|
|
except ValueError:
|
|
continue
|
|
return False
|
|
|
|
def _under_writable_root(self, path: str) -> bool:
|
|
candidate = self._candidate(path)
|
|
for rp, writable in self._resolved_roots():
|
|
if not writable:
|
|
continue
|
|
try:
|
|
candidate.relative_to(rp)
|
|
return True
|
|
except ValueError:
|
|
continue
|
|
return False
|
|
|
|
def _domain_allowed(self, url: str) -> bool:
|
|
"""True when the URL's host is an allowed egress destination — an exact match or a
|
|
subdomain of an allowed domain (so `docs.python.org` matches `python.org`, but
|
|
`evil-python.org` never matches `python.org`)."""
|
|
host = _host_of(url)
|
|
if not host:
|
|
return False
|
|
allowed = {d for d in (_host_of(x) for x in self.allowed_domains) if d}
|
|
allowed |= self.session_allow_domains
|
|
for dom in allowed:
|
|
if host == dom or host.endswith("." + dom):
|
|
return True
|
|
return False
|
|
|
|
def _command_allowed(self, command: str) -> bool:
|
|
# An allowlist entry auto-runs a command WITHOUT approval, so prefix matching is
|
|
# unsafe: `git status` would auto-approve `git status && rm -rf ~`. Reject anything
|
|
# carrying shell operators (chaining/redirection/substitution) up front, then match
|
|
# the parsed argv against each entry — the entry's own tokens must be an exact
|
|
# prefix of the command's tokens (so `git status` matches `git status -s` but never
|
|
# `git statusfoo` or a bare `git`).
|
|
if _has_shell_operators(command):
|
|
return False
|
|
try:
|
|
argv = shlex.split(command)
|
|
except ValueError:
|
|
return False # unbalanced quotes etc. — treat as not-allowlisted
|
|
if not argv:
|
|
return False
|
|
for allowed in self.allowed_commands:
|
|
try:
|
|
prefix = shlex.split(allowed)
|
|
except ValueError:
|
|
continue
|
|
if prefix and argv[: len(prefix)] == prefix:
|
|
return True
|
|
return False
|