Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
9 changes: 9 additions & 0 deletions .gitignore
Original file line number Diff line number Diff line change
Expand Up @@ -48,3 +48,12 @@ node_modules
# so it stays on disk and is only untracked. Deliberately unanchored: an install
# run from web/ leaves the same store one level down.
.pnpm-store/

# otari hook setup's own generated settings: both embed a live Otari API key
# or master key directly in the hook command (see docs/agent-gates.md), so
# neither is safe to commit. Only these two files, not the whole .claude/ or
# .codex/ directory: .claude/settings.json (team-shared, committed on
# purpose) and any .claude/skills/ live under the same directory and must
# stay tracked.
.claude/settings.local.json
.codex/hooks.json
222 changes: 173 additions & 49 deletions docs/agent-gates.md

Large diffs are not rendered by default.

36 changes: 35 additions & 1 deletion src/gateway/agent_runtime/domain/policy.py
Original file line number Diff line number Diff line change
Expand Up @@ -55,6 +55,14 @@
# time rather than a gate that silently blocks on a model's say-so.
_JUDGE_ENFORCEMENTS = {"advisory"}

# Which locally-installed CLI(s) `otari hook` may use for a judge gate's own
# model call (JudgeGate.judge_cli); see that field's own docstring. Kept as
# its own set, not reused from anywhere `otari hook` itself defines, since
# this module stays dependency-free of that CLI-only concern (subprocess
# names, PATH resolution): a gate author only ever needs to know these two
# names exist, not how either is actually invoked.
_SUPPORTED_JUDGE_CLIS = {"claude", "codex"}

# A `**` in a forbidden glob crosses path segments by recursing over every
# split point in the submitted path (domain/evaluators.py's _segments_match).
# One is what every example in this codebase uses; more than that multiplies
Expand All @@ -75,7 +83,7 @@
"changed_path": _COMMON_GATE_FIELDS | {"forbidden"},
"command_match": _COMMON_GATE_FIELDS | {"forbidden"},
"command_if_changed": _COMMON_GATE_FIELDS | {"when_changed", "require"},
"judge": _COMMON_GATE_FIELDS | {"rubric", "when_changed"},
"judge": _COMMON_GATE_FIELDS | {"rubric", "when_changed", "judge_cli"},
"check_passed": _COMMON_GATE_FIELDS | {"verifier", "when_changed"},
}

Expand Down Expand Up @@ -320,6 +328,31 @@ def _parse_gate(raw: Any) -> GateSpec:
judge_when_changed = _parse_glob_list(
gate_id, "when_changed", _require_string_list(raw, "when_changed", gate_id, gate_type)
)
# judge_cli: optional, like when_changed above; absence means "no
# preference" (see JudgeGate's own docstring), not "always these two".
# Accepts a bare string as the one-entry case of the list form, not a
# different shape, since a gate author naming a single required CLI
# should not have to spell it as a one-item list.
judge_cli: tuple[str, ...] | None = None
if "judge_cli" in raw:
raw_judge_cli = raw["judge_cli"]
if isinstance(raw_judge_cli, str):
candidates = [raw_judge_cli] if raw_judge_cli else []
elif isinstance(raw_judge_cli, list) and all(isinstance(item, str) for item in raw_judge_cli):
candidates = list(dict.fromkeys(raw_judge_cli))
else:
candidates = []
if not candidates:
raise PolicyError(
f"Gate {gate_id!r} (type 'judge'): 'judge_cli' must be a non-empty string or list of strings."
)
unsupported = [item for item in candidates if item not in _SUPPORTED_JUDGE_CLIS]
if unsupported:
raise PolicyError(
f"Gate {gate_id!r}: judge_cli entries {unsupported!r} are not supported. "
f"Supported in this build: {', '.join(sorted(_SUPPORTED_JUDGE_CLIS))}."
)
judge_cli = tuple(candidates)
# enforcement_value is already proven "advisory" by the _JUDGE_ENFORCEMENTS
# check above; cast documents that narrowing the same way the plain
# Enforcement cast above documents its own.
Expand All @@ -328,6 +361,7 @@ def _parse_gate(raw: Any) -> GateSpec:
enforcement=cast(Literal["advisory"], enforcement_value),
rubric=rubric,
when_changed=tuple(judge_when_changed),
judge_cli=judge_cli,
message=message,
)

Expand Down
14 changes: 14 additions & 0 deletions src/gateway/agent_runtime/domain/types.py
Original file line number Diff line number Diff line change
Expand Up @@ -154,13 +154,27 @@ class JudgeGate:
touched a matching path, so a rubric about, say, error-handling
conventions is not re-judged, at real model-call cost, on a session that
never touched application code.

``judge_cli`` is optional and names which locally-installed CLI(s)
``otari hook`` may use to make the model call this gate needs, in
preference order; the first one whose own binary is found on ``PATH``
wins. ``None`` (the default, and the only behavior a judge gate had
before this field existed) means no preference: the caller falls back to
whichever CLI its own invoking harness implies (Claude Code's hook ->
``claude``, Codex's -> ``codex``). Naming one explicitly is what lets a
gate authored for, say, a Codex-only fleet require ``codex`` even when
invoked by a Claude Code hook, or list both so whichever is actually
installed on a given machine is used. This field changes nothing about
where the call happens: still entirely within ``otari hook``, never here
(see this gate's own opening paragraph).
"""

id: str
enforcement: Literal["advisory"]
rubric: str
message: str
when_changed: tuple[str, ...] = ()
judge_cli: tuple[str, ...] | None = None
type: Literal["judge"] = "judge"


Expand Down
Loading
Loading