From 7ce35ab7c83026660739ea321c58ff16301387be Mon Sep 17 00:00:00 2001 From: Bai Li Date: Sat, 3 Oct 2026 09:08:49 -0700 Subject: [PATCH 1/5] fix(antigravity): keep schedule enabled under a tool allowlist The allowlist from #215 maps Claude tool names onto Antigravity builtins, so `schedule` (no Claude equivalent) was always dropped. Gemini uses it to sleep while a backgrounded run_command finishes; without it, it re-checks the task every few seconds and each check counts against max_turns. Co-Authored-By: Claude Opus 5.5 --- .claude/notes/agents.md | 7 ++++++- src/coder_eval/agents/antigravity_agent.py | 7 ++++--- tests/test_antigravity_agent.py | 17 ++++++++++++----- 3 files changed, 22 insertions(+), 9 deletions(-) diff --git a/.claude/notes/agents.md b/.claude/notes/agents.md index 74f486bc..dc4413d4 100644 --- a/.claude/notes/agents.md +++ b/.claude/notes/agents.md @@ -464,7 +464,12 @@ after the model had already finished. `finalize` force-closes them as unresolved way it confines Claude Code. Without it Antigravity ran with every builtin, including `start_subagent` and `search_web`, under a list that gave Claude Code neither. `Skill` and `TodoWrite` have no builtin (skills load through `skills_paths`) and are skipped. `finish` -stays on under any allowlist: it returns structured output, not a capability. +and `schedule` stay on under any allowlist: `finish` returns structured output, and +`schedule` is the timer the model sleeps on while a backgrounded `run_command` finishes. +Without `schedule`, Gemini re-checks the background task every few seconds and each check +is a model call against `max_turns`: on the SkillsBench Gemini 4-arm campaign mean turns +rose 36.8 → 50.2 and max-turn rows 4 → 16 on the same 110 rows, and the skill arms were hit +2-3x harder than baseline. With `schedule` on they came back to 35.4 and 1. `enable_subagents` is a separate switch from the toolset and follows whether `start_subagent` survives. The SDK takes an allowlist OR a denylist, so with both set the denied tools are removed from the allowlist. diff --git a/src/coder_eval/agents/antigravity_agent.py b/src/coder_eval/agents/antigravity_agent.py index 3f28cb78..262ea26a 100644 --- a/src/coder_eval/agents/antigravity_agent.py +++ b/src/coder_eval/agents/antigravity_agent.py @@ -148,9 +148,10 @@ "AskUserQuestion": "ask_question", } -# Kept on under any allowlist: `finish` is how a turn returns structured -# output, not a capability an allowlist is meant to grant or withhold. -_ALWAYS_ENABLED_TOOLS: frozenset[str] = frozenset({"finish"}) +# Kept on under any allowlist: `finish` returns a turn's structured output and +# `schedule` is how the model waits on a backgrounded command. Neither is a +# capability an allowlist is meant to grant or withhold. +_ALWAYS_ENABLED_TOOLS: frozenset[str] = frozenset({"finish", "schedule"}) # Set on every run_command unless the environment already sets them, so a # command that would stop to ask (`npx` installing a package, git credentials, diff --git a/tests/test_antigravity_agent.py b/tests/test_antigravity_agent.py index c5ddfebe..5be4d1bd 100644 --- a/tests/test_antigravity_agent.py +++ b/tests/test_antigravity_agent.py @@ -1536,7 +1536,7 @@ def test_tool_capabilities_none_without_tool_lists(): def test_allowed_tools_map_to_enabled_builtins(): - """The default experiment allowlist enables exactly the matching builtins, plus `finish`. + """The default experiment allowlist enables exactly the matching builtins, plus `finish` and `schedule`. `Skill` has no builtin (skills load through skills_paths) and is skipped; subagents, web search and the rest stay off, as they do for Claude Code under the same list. @@ -1549,6 +1549,7 @@ def test_allowed_tools_map_to_enabled_builtins(): "find_file", "finish", "run_command", + "schedule", "search_directory", "view_file", ] @@ -1558,7 +1559,7 @@ def test_allowed_tools_map_to_enabled_builtins(): def test_allowed_task_keeps_subagents(): caps = _capabilities(allowed_tools=["Bash", "Task"]) - assert {t.value for t in caps.enabled_tools} == {"run_command", "start_subagent", "finish"} + assert {t.value for t in caps.enabled_tools} == {"run_command", "start_subagent", "finish", "schedule"} assert caps.enable_subagents is True @@ -1573,7 +1574,7 @@ def test_disallowed_tools_are_removed_from_the_allowlist(): """The SDK takes an allowlist OR a denylist, so with both the denied tools leave the allowlist.""" caps = _capabilities(allowed_tools=["Bash", "Read", "Task"], disallowed_tools=["Task"]) - assert {t.value for t in caps.enabled_tools} == {"run_command", "view_file", "finish"} + assert {t.value for t in caps.enabled_tools} == {"run_command", "view_file", "finish", "schedule"} assert caps.enable_subagents is False @@ -1594,7 +1595,12 @@ async def __aexit__(self, *exc): await _agent(allowed_tools=["Bash", "Read"]).start(str(tmp_path)) - assert {t.value for t in configs[0].capabilities.enabled_tools} == {"run_command", "view_file", "finish"} + assert {t.value for t in configs[0].capabilities.enabled_tools} == { + "run_command", + "view_file", + "finish", + "schedule", + } def test_installed_sdk_accepts_the_tool_capabilities(): @@ -1602,7 +1608,8 @@ def test_installed_sdk_accepts_the_tool_capabilities(): every mapped name is a real builtin, so a renamed tool fails here, not live.""" types = pytest.importorskip("google.antigravity").types - assert set(agent_module._CLAUDE_TO_ANTIGRAVITY_TOOL_MAP.values()) <= {t.value for t in types.BuiltinTools} + mapped = set(agent_module._CLAUDE_TO_ANTIGRAVITY_TOOL_MAP.values()) | agent_module._ALWAYS_ENABLED_TOOLS + assert mapped <= {t.value for t in types.BuiltinTools} caps = _agent(allowed_tools=["Bash", "Read", "Write", "Edit", "Glob", "Grep", "Skill"])._tool_capabilities(types) assert isinstance(caps, types.CapabilitiesConfig) assert types.BuiltinTools.START_SUBAGENT not in caps.enabled_tools From eaea30d9061f39c824ac8fb2662f883abc01d0ac Mon Sep 17 00:00:00 2001 From: Bai Li Date: Sat, 3 Oct 2026 09:38:29 -0700 Subject: [PATCH 2/5] fix(antigravity): keep a tool allowlist within the harness default toolset The allowlist mapped Glob/Grep/LS onto find_file/search_directory/ list_directory, which the harness ships off (BuiltinTools.deprecated()). Turning them on changed how Gemini works: on SkillsBench it made 1,429 Glob/Grep calls in 333 of 348 rows where 0.12.9 made none, often one search per turn in place of a script. Co-Authored-By: Claude Opus 5.5 --- .claude/notes/agents.md | 6 ++++++ src/coder_eval/agents/antigravity_agent.py | 11 +++++++---- tests/test_antigravity_agent.py | 21 ++++++++++++++++----- 3 files changed, 29 insertions(+), 9 deletions(-) diff --git a/.claude/notes/agents.md b/.claude/notes/agents.md index dc4413d4..09ad3fd3 100644 --- a/.claude/notes/agents.md +++ b/.claude/notes/agents.md @@ -470,6 +470,12 @@ Without `schedule`, Gemini re-checks the background task every few seconds and e is a model call against `max_turns`: on the SkillsBench Gemini 4-arm campaign mean turns rose 36.8 → 50.2 and max-turn rows 4 → 16 on the same 110 rows, and the skill arms were hit 2-3x harder than baseline. With `schedule` on they came back to 35.4 and 1. + +An allowlist only narrows the harness's default toolset (`BuiltinTools.default()`). `Glob`, +`Grep` and `LS` map to `find_file`, `search_directory` and `list_directory`, which the +harness ships off, so an allowlist naming them leaves them off. Turned on, Gemini used them +in place of shell scripts (1,429 calls in 333 of 348 SkillsBench rows, none before), often +one search per turn. `enable_subagents` is a separate switch from the toolset and follows whether `start_subagent` survives. The SDK takes an allowlist OR a denylist, so with both set the denied tools are removed from the allowlist. diff --git a/src/coder_eval/agents/antigravity_agent.py b/src/coder_eval/agents/antigravity_agent.py index 262ea26a..ece20ccd 100644 --- a/src/coder_eval/agents/antigravity_agent.py +++ b/src/coder_eval/agents/antigravity_agent.py @@ -371,9 +371,11 @@ def _tool_capabilities(self, types: Any) -> Any: ``allowed_tools`` / ``disallowed_tools``, or ``None`` when neither is set. Names are mapped through ``_CLAUDE_TO_ANTIGRAVITY_TOOL_MAP``; names with no - Antigravity builtin (``Skill``, ``TodoWrite``, MCP tools) are skipped. The - SDK takes an allowlist OR a denylist, so with both set the denied tools are - removed from the allowlist. + Antigravity builtin (``Skill``, ``TodoWrite``, MCP tools) are skipped. An + allowlist only narrows the harness's default toolset, so ``Glob`` / ``Grep`` / + ``LS`` never turn on the tools the harness ships off (``find_file``, + ``search_directory``, ``list_directory``). The SDK takes an allowlist OR a + denylist, so with both set the denied tools are removed from the allowlist. Rationale: .claude/notes/agents.md § Antigravity tool allowlist """ @@ -381,6 +383,7 @@ def _tool_capabilities(self, types: Any) -> Any: if not allowed and not disallowed: return None builtin = {t.value for t in types.BuiltinTools} + harness_default = {t.value for t in types.BuiltinTools.default()} def to_builtin(names: list[str] | None) -> set[str]: mapped = {_CLAUDE_TO_ANTIGRAVITY_TOOL_MAP.get(n, n) for n in names or []} @@ -390,7 +393,7 @@ def to_builtin(names: list[str] | None) -> set[str]: # whether `start_subagent` survives the filter. subagent = types.BuiltinTools.START_SUBAGENT.value if allowed: - enabled = (to_builtin(allowed) | _ALWAYS_ENABLED_TOOLS) - to_builtin(disallowed) + enabled = ((to_builtin(allowed) & harness_default) | _ALWAYS_ENABLED_TOOLS) - to_builtin(disallowed) self._log.debug("Enabled builtin tools: %s", ", ".join(sorted(enabled))) return types.CapabilitiesConfig( enabled_tools=[types.BuiltinTools(t) for t in sorted(enabled)], diff --git a/tests/test_antigravity_agent.py b/tests/test_antigravity_agent.py index 5be4d1bd..b43ce3b8 100644 --- a/tests/test_antigravity_agent.py +++ b/tests/test_antigravity_agent.py @@ -508,6 +508,11 @@ class _FakeBuiltinTools(enum.StrEnum): SCHEDULE = "schedule" FINISH = "finish" + @classmethod + def default(cls) -> list["_FakeBuiltinTools"]: + off = {cls.ASK_QUESTION, cls.LIST_DIR, cls.SEARCH_DIR, cls.FIND_FILE} + return [t for t in cls if t not in off] + def _install_fake_sdk(monkeypatch, sdk_agent_cls) -> None: """Stub ``google.antigravity`` in sys.modules so ``start()`` runs without the extra. @@ -1536,26 +1541,31 @@ def test_tool_capabilities_none_without_tool_lists(): def test_allowed_tools_map_to_enabled_builtins(): - """The default experiment allowlist enables exactly the matching builtins, plus `finish` and `schedule`. + """The default experiment allowlist enables exactly the matching default builtins, plus `finish` and `schedule`. - `Skill` has no builtin (skills load through skills_paths) and is skipped; subagents, - web search and the rest stay off, as they do for Claude Code under the same list. + `Skill` has no builtin (skills load through skills_paths) and is skipped; `Glob` / `Grep` + map to tools the harness ships off, so they stay off; subagents, web search and the rest + stay off, as they do for Claude Code under the same list. """ caps = _capabilities(allowed_tools=["Bash", "Read", "Write", "Edit", "Glob", "Grep", "Skill"]) assert [t.value for t in caps.enabled_tools] == [ "create_file", "edit_file", - "find_file", "finish", "run_command", "schedule", - "search_directory", "view_file", ] assert caps.enable_subagents is False +def test_allowlist_never_enables_tools_the_harness_ships_off(): + caps = _capabilities(allowed_tools=["Bash", "Glob", "Grep", "LS", "AskUserQuestion"]) + + assert {t.value for t in caps.enabled_tools} == {"run_command", "finish", "schedule"} + + def test_allowed_task_keeps_subagents(): caps = _capabilities(allowed_tools=["Bash", "Task"]) @@ -1613,6 +1623,7 @@ def test_installed_sdk_accepts_the_tool_capabilities(): caps = _agent(allowed_tools=["Bash", "Read", "Write", "Edit", "Glob", "Grep", "Skill"])._tool_capabilities(types) assert isinstance(caps, types.CapabilitiesConfig) assert types.BuiltinTools.START_SUBAGENT not in caps.enabled_tools + assert types.BuiltinTools.FIND_FILE not in caps.enabled_tools # --- max_turns cap ------------------------------------------------------------------- From e794947ce2cfd18c1b982cf460ba77905190b71f Mon Sep 17 00:00:00 2001 From: Bai Li Date: Sat, 3 Oct 2026 10:45:19 -0700 Subject: [PATCH 3/5] fix(antigravity): bound the background poll by the deadline alone when a timeout is set The 120-cycle (10 minute) cap applied alongside the 80%-of-turn_timeout deadline, so a slow solver or simulation was force-closed at 10 minutes even with turn budget left. On the SkillsBench Gemini 4-arm campaign it ended 22 of 348 rows; 4 had passed on 0.12.9 with the deadline alone, e.g. an exam-scheduling MIP solve that finished at 1211s. The cap is again the bound only when a task sets no timeout. Co-Authored-By: Claude Opus 5.5 --- .claude/notes/agents.md | 13 +++++--- docs/agents/ANTIGRAVITY.md | 4 +-- src/coder_eval/agents/antigravity_agent.py | 21 +++++++------ tests/test_antigravity_agent.py | 36 ++++++++++++++-------- 4 files changed, 45 insertions(+), 29 deletions(-) diff --git a/.claude/notes/agents.md b/.claude/notes/agents.md index 09ad3fd3..169a6eaf 100644 --- a/.claude/notes/agents.md +++ b/.claude/notes/agents.md @@ -434,10 +434,13 @@ agent had produced; bounding it only by a fixed cycle count disconnected from `t path unreachable — the watchdog always wins, and the same spurious-orphan turn burns the full turn timeout before crashing with zero criteria evaluated. -The cycle cap applies alongside that deadline, whichever comes first, and is the SOLE -bound when a task sets no timeout at all. Under a long `turn_timeout` (1800 s gives a -1440 s deadline) the deadline alone let a never-ending command (a dev server, an -unanswered prompt) idle the turn for 24 minutes. It is deliberately not +The cycle cap (120 × 5 s) bounds the wait only when a task sets no timeout at all. With +one, the deadline alone bounds it, so a never-ending command (a dev server, an unanswered +prompt) can idle up to 80% of the turn, but a slow job gets the time the task author +budgeted for it. Applying the cap under a timeout as well force-closed real solvers and +simulations at 10 minutes: on the SkillsBench Gemini 4-arm campaign it ended 22 of 348 +rows, 4 of which had passed with the deadline alone (an exam-scheduling MIP solve that +passed at 1211 s among them). It is deliberately not "break after N consecutive empty polls": `receive_steps()` returns identically empty whether a backgrounded job is still running or will never resolve, and there is no signal that tells the two apart except waiting. A count small enough to matter would abort real @@ -489,7 +492,7 @@ behind keeps the turn waiting on a prompt no one answers. `_NONINTERACTIVE_ENV` (`CI`, `npm_config_yes`, `GIT_TERMINAL_PROMPT=0`, `DEBIAN_FRONTEND`, `PIP_NO_INPUT`, `PAGER`/`GIT_PAGER=cat`) rides the per-agent `env` seam, each variable only where the environment does not already set it, so those commands answer themselves or fail fast. It -cannot close the terminal: a bare `read` or a server still runs until the poll cap. +cannot close the terminal: a bare `read` or a server still runs until the poll deadline. ## The receive_steps re-entrancy window diff --git a/docs/agents/ANTIGRAVITY.md b/docs/agents/ANTIGRAVITY.md index ad6aefba..b0f152e2 100644 --- a/docs/agents/ANTIGRAVITY.md +++ b/docs/agents/ANTIGRAVITY.md @@ -205,8 +205,8 @@ as every other agent. 10-second maximum synchronous wait; past it the command becomes a background task and the model gets a task id, not a result. The turn polls for that result instead of finalizing on an idle step stream, so slow work does complete — but only an - orphaned `run_command` is waited on, the wait is bounded by 10 minutes or 80% of - `turn_timeout` (whichever is shorter), and a job that outlives it (typically a server + orphaned `run_command` is waited on, the wait is bounded by 80% of `turn_timeout` + (10 minutes when the task sets none), and a job that outlives it (typically a server the model left running) is force-closed as `result_status: unknown` and graded normally rather than as a timeout. Measured in [Run-Limit Parity](HARNESS_PARITY.md). diff --git a/src/coder_eval/agents/antigravity_agent.py b/src/coder_eval/agents/antigravity_agent.py index ece20ccd..f060df7b 100644 --- a/src/coder_eval/agents/antigravity_agent.py +++ b/src/coder_eval/agents/antigravity_agent.py @@ -95,11 +95,9 @@ # Rationale: .claude/notes/agents.md § Antigravity Step interleaving and the background poll _POLL_DEADLINE_TIMEOUT_FRACTION = 0.8 -# Cap on poll *cycles* -- the SOLE bound when a task sets no timeout at all, and -# a backstop against a very large one (applied alongside the deadline, whichever -# is reached first). 120 * 5s = 10 minutes, ~2x the worst real -# backgrounded-job duration observed (60-300s). Deliberately NOT "break after N -# consecutive empty polls". +# Cap on poll *cycles*, the bound only when a task sets no timeout at all; with +# one, the deadline alone bounds the wait. 120 * 5s = 10 minutes. Deliberately +# NOT "break after N consecutive empty polls". # Rationale: .claude/notes/agents.md § Antigravity Step interleaving and the background poll _MAX_BACKGROUND_POLLS = 120 @@ -639,8 +637,11 @@ def _on_turn_timeout() -> None: and not state.max_turns_hit and not state.timeout_hit and state.has_orphaned_tool_call() - and poll_count < _MAX_BACKGROUND_POLLS - and (poll_deadline is None or time.monotonic() < poll_deadline) + and ( + poll_count < _MAX_BACKGROUND_POLLS + if poll_deadline is None + else time.monotonic() < poll_deadline + ) ): poll_count += 1 self._log.debug("Polling for backgrounded work (orphaned tool call); attempt %d", poll_count) @@ -666,9 +667,9 @@ def _on_turn_timeout() -> None: # stop/timeout: the call is force-closed as unresolved and # the turn is still graded normally on everything else. bound = ( - f"_MAX_BACKGROUND_POLLS ({_MAX_BACKGROUND_POLLS})" - if poll_count >= _MAX_BACKGROUND_POLLS - else f"poll_deadline ({_POLL_DEADLINE_TIMEOUT_FRACTION:.0%} of {timeout:g}s turn timeout)" + f"poll_deadline ({_POLL_DEADLINE_TIMEOUT_FRACTION:.0%} of {timeout:g}s turn timeout)" + if poll_deadline is not None + else f"_MAX_BACKGROUND_POLLS ({_MAX_BACKGROUND_POLLS})" ) msg = "Poll budget exhausted (%s, poll_count=%d) with a tool call still ACTIVE." self._log.warning(msg, bound, poll_count) diff --git a/tests/test_antigravity_agent.py b/tests/test_antigravity_agent.py index b43ce3b8..9c3e8fcd 100644 --- a/tests/test_antigravity_agent.py +++ b/tests/test_antigravity_agent.py @@ -888,10 +888,10 @@ async def _record_sleep(seconds: float) -> None: assert bash.result_status == "unknown" # force-closed as UNRESOLVED by finalize() -async def test_communicate_poll_cap_also_bounds_a_turn_with_a_large_timeout(monkeypatch): - """The cycle cap applies alongside the timeout-derived deadline, not only when - no timeout is set: under a large turn_timeout (1800s gives a 1440s deadline) a - never-closing job stops at _MAX_BACKGROUND_POLLS instead of the deadline.""" +async def test_communicate_poll_cap_does_not_cut_short_a_job_inside_the_deadline(monkeypatch): + """With a turn_timeout the deadline alone bounds the wait: a solver still running + past _MAX_BACKGROUND_POLLS cycles is waited on until it finishes. Seen live: a MIP + solve that passed at 1211s was force-closed at the 10-minute cap.""" from coder_eval.agents import antigravity_agent monkeypatch.setattr(antigravity_agent, "_MAX_BACKGROUND_POLLS", 3) @@ -902,22 +902,34 @@ async def _record_sleep(seconds: float) -> None: monkeypatch.setattr(antigravity_agent.asyncio, "sleep", _record_sleep) - never_closing = [ + started = [ _step( "TOOL_CALL", "ACTIVE", target="TARGET_ENVIRONMENT", - tool_calls=[_tc("run_command", "stuck", {"command_line": "node server.js"})], + tool_calls=[_tc("run_command", "solve", {"command_line": "python solve.py"})], ), - _step("TEXT_RESPONSE", "DONE", content="server started", complete=True, usage=_usage(10, 0, 1, 0)), + _step("TEXT_RESPONSE", "DONE", content="solver running", complete=True, usage=_usage(10, 0, 1, 0)), ] - agent = _agent_with_steps([never_closing]) - tr = await agent.communicate("start it", timeout=1800.0) + finished = [ + _step( + "TOOL_CALL", + "DONE", + target="TARGET_ENVIRONMENT", + tool_calls=[ + _tc( + "run_command", "solve", {"command_line": "python solve.py", "exit_code": 0, "combined_output": "ok"} + ) + ], + ), + _step("TEXT_RESPONSE", "DONE", content="solved", complete=True, usage=_usage(10, 0, 1, 0)), + ] + agent = _agent_with_steps([started, [], [], [], [], finished]) + tr = await agent.communicate("solve it", timeout=1800.0) - assert len(sleep_calls) == 3 # the cap, long before the 1440s deadline + assert len(sleep_calls) == 5 bash = next(c for c in tr.commands if c.tool_name == "Bash") - assert bash.result_status == "unknown" - assert tr.agent_output == "server started" + assert bash.result_status == "success" async def test_communicate_does_not_poll_a_non_command_tool_left_active(monkeypatch): From 481fcb1fdb4499a8367288e0e965adf07e68ca25 Mon Sep 17 00:00:00 2001 From: Bai Li Date: Sat, 3 Oct 2026 11:48:45 -0700 Subject: [PATCH 4/5] fix(antigravity): keep start_subagent on and deny withheld subagents by policy The harness builds its system prompt from the toolset. With start_subagent off it drops the whole subagents section, which carries the prompt's only "you do NOT need to poll, you will be notified" guidance. Under an allowlist without Task, Gemini then polls backgrounded commands with status checks and short liveness timers: 1% of rows on 0.12.9 vs 8-10% after #215, and max-turn rows 2 vs 12 of 148 on the SkillsBench Gemini rerun. start_subagent now stays on, so the prompt matches the harness default, and when the tool lists do not allow Task the call that runs a subagent (invoke_subagent) is denied by policy. That covers the built-in research subagent and model-defined ones, which both get search_web regardless of the parent's toolset. Co-Authored-By: Claude Opus 5.5 --- .claude/notes/agents.md | 16 ++++-- docs/agents/ANTIGRAVITY.md | 4 +- src/coder_eval/agents/antigravity_agent.py | 43 ++++++++------- tests/test_antigravity_agent.py | 62 +++++++++++++++------- 4 files changed, 85 insertions(+), 40 deletions(-) diff --git a/.claude/notes/agents.md b/.claude/notes/agents.md index 169a6eaf..3b963508 100644 --- a/.claude/notes/agents.md +++ b/.claude/notes/agents.md @@ -479,9 +479,19 @@ An allowlist only narrows the harness's default toolset (`BuiltinTools.default() harness ships off, so an allowlist naming them leaves them off. Turned on, Gemini used them in place of shell scripts (1,429 calls in 333 of 348 SkillsBench rows, none before), often one search per turn. -`enable_subagents` is a separate switch from the toolset and follows whether -`start_subagent` survives. The SDK takes an allowlist OR a denylist, so with both set the -denied tools are removed from the allowlist. + +`start_subagent` also stays on under any allowlist, because the harness builds its system +prompt from the toolset: with `start_subagent` off it drops the whole subagents section +(4,801 → 2,783 characters), and with it the prompt's only "you do NOT need to poll ... you +will be notified" guidance. Without that, Gemini polls a backgrounded command with status +checks and short "liveness" timers: rows doing so rose from 1% (0.12.9) to 8-10%, and +max-turn rows from 2 to 12 of 148. The tool descriptions are identical either way, and +`enable_subagents` alone does not restore the section. Subagents the tool lists withhold +(no `Task`) are denied by policy at `invoke_subagent` instead, which covers both the +built-in `research` subagent and one the model defines with `define_subagent`; both get +`search_web` regardless of the parent's toolset, and a parent policy on `search_web` does +not reach a model-defined one. The SDK takes an allowlist OR a denylist, so with both set +the denied tools are removed from the allowlist, except the always-enabled ones. ## Antigravity non-interactive commands diff --git a/docs/agents/ANTIGRAVITY.md b/docs/agents/ANTIGRAVITY.md index b0f152e2..a44f4378 100644 --- a/docs/agents/ANTIGRAVITY.md +++ b/docs/agents/ANTIGRAVITY.md @@ -145,7 +145,9 @@ sandbox working directory plus any skill roots). onto the harness's builtin tools (`Bash` → `run_command`, `Read` → `view_file`, `Write` → `create_file`, `Edit` → `edit_file`, `Glob` → `find_file`, `Grep` → `search_directory`, `Task` → `start_subagent`, `WebSearch` → `search_web`, `WebFetch` → `read_url_content`). -`Skill` has no builtin and is skipped; `finish` always stays on. +`Skill` has no builtin and is skipped. `finish`, `schedule` and `start_subagent` always +stay on, the last so the harness keeps its default system prompt; when the lists do not +allow `Task`, running a subagent (`invoke_subagent`) is denied by policy instead. `run_command` also gets non-interactive environment variables (`CI=1`, `npm_config_yes=true`, `GIT_TERMINAL_PROMPT=0`, `DEBIAN_FRONTEND=noninteractive`, diff --git a/src/coder_eval/agents/antigravity_agent.py b/src/coder_eval/agents/antigravity_agent.py index f060df7b..1f00afd1 100644 --- a/src/coder_eval/agents/antigravity_agent.py +++ b/src/coder_eval/agents/antigravity_agent.py @@ -146,10 +146,15 @@ "AskUserQuestion": "ask_question", } -# Kept on under any allowlist: `finish` returns a turn's structured output and -# `schedule` is how the model waits on a backgrounded command. Neither is a -# capability an allowlist is meant to grant or withhold. -_ALWAYS_ENABLED_TOOLS: frozenset[str] = frozenset({"finish", "schedule"}) +# Kept on under any allowlist: `finish` returns a turn's structured output, +# `schedule` is how the model waits on a backgrounded command, and the harness +# drops its "you will be notified, do not poll" guidance with `start_subagent`. +# Withheld subagents are denied at `_SUBAGENT_CALL` instead. +# Rationale: .claude/notes/agents.md § Antigravity tool allowlist +_ALWAYS_ENABLED_TOOLS: frozenset[str] = frozenset({"finish", "schedule", "start_subagent"}) + +# The one call that runs a subagent, built-in or model-defined. +_SUBAGENT_CALL = "invoke_subagent" # Set on every run_command unless the environment already sets them, so a # command that would stop to ask (`npx` installing a package, git credentials, @@ -374,6 +379,7 @@ def _tool_capabilities(self, types: Any) -> Any: ``LS`` never turn on the tools the harness ships off (``find_file``, ``search_directory``, ``list_directory``). The SDK takes an allowlist OR a denylist, so with both set the denied tools are removed from the allowlist. + ``_ALWAYS_ENABLED_TOOLS`` are never removed. Rationale: .claude/notes/agents.md § Antigravity tool allowlist """ @@ -387,22 +393,22 @@ def to_builtin(names: list[str] | None) -> set[str]: mapped = {_CLAUDE_TO_ANTIGRAVITY_TOOL_MAP.get(n, n) for n in names or []} return mapped & builtin - # `enable_subagents` is a separate switch from the toolset; it follows - # whether `start_subagent` survives the filter. - subagent = types.BuiltinTools.START_SUBAGENT.value if allowed: - enabled = ((to_builtin(allowed) & harness_default) | _ALWAYS_ENABLED_TOOLS) - to_builtin(disallowed) + enabled = ((to_builtin(allowed) & harness_default) - to_builtin(disallowed)) | _ALWAYS_ENABLED_TOOLS self._log.debug("Enabled builtin tools: %s", ", ".join(sorted(enabled))) - return types.CapabilitiesConfig( - enabled_tools=[types.BuiltinTools(t) for t in sorted(enabled)], - enable_subagents=subagent in enabled, - ) + return types.CapabilitiesConfig(enabled_tools=[types.BuiltinTools(t) for t in sorted(enabled)]) disabled = to_builtin(disallowed) - _ALWAYS_ENABLED_TOOLS self._log.debug("Disabled builtin tools: %s", ", ".join(sorted(disabled))) - return types.CapabilitiesConfig( - disabled_tools=[types.BuiltinTools(t) for t in sorted(disabled)], - enable_subagents=subagent not in disabled, - ) + return types.CapabilitiesConfig(disabled_tools=[types.BuiltinTools(t) for t in sorted(disabled)]) + + def _subagents_allowed(self) -> bool: + """Whether ``allowed_tools`` / ``disallowed_tools`` let the model run subagents (``Task``).""" + + def names_task(names: list[str] | None) -> bool: + return any(_CLAUDE_TO_ANTIGRAVITY_TOOL_MAP.get(n, n) == "start_subagent" for n in names or []) + + allowed, disallowed = self.config.allowed_tools, self.config.disallowed_tools + return (not allowed or names_task(allowed)) and not names_task(disallowed) async def start( self, @@ -453,8 +459,9 @@ async def start( # policy would deny. ``permission_mode`` is deliberately NOT mapped # here — it does not confine this agent, exactly as on Codex, and # docs/agents/HARNESS_PARITY.md says so rather than leaving it - # silent. The isolation boundary is the driver. - policies=[policy.allow_all()], + # silent. The isolation boundary is the driver. Subagents the tool + # lists withhold are denied here, since `start_subagent` stays on. + policies=[policy.allow_all()] + ([] if self._subagents_allowed() else [policy.deny(_SUBAGENT_CALL)]), system_instructions=self.config.system_prompt or None, # Skill discovery: the search-path roots that parent the skill dirs. skills_paths=skills_paths, diff --git a/tests/test_antigravity_agent.py b/tests/test_antigravity_agent.py index 9c3e8fcd..160f34fe 100644 --- a/tests/test_antigravity_agent.py +++ b/tests/test_antigravity_agent.py @@ -1553,11 +1553,12 @@ def test_tool_capabilities_none_without_tool_lists(): def test_allowed_tools_map_to_enabled_builtins(): - """The default experiment allowlist enables exactly the matching default builtins, plus `finish` and `schedule`. + """The default experiment allowlist enables exactly the matching default builtins, plus + `finish`, `schedule` and `start_subagent`. `Skill` has no builtin (skills load through skills_paths) and is skipped; `Glob` / `Grep` - map to tools the harness ships off, so they stay off; subagents, web search and the rest - stay off, as they do for Claude Code under the same list. + map to tools the harness ships off, so they stay off; web search and the rest stay off, as + they do for Claude Code under the same list. """ caps = _capabilities(allowed_tools=["Bash", "Read", "Write", "Edit", "Glob", "Grep", "Skill"]) @@ -1567,37 +1568,61 @@ def test_allowed_tools_map_to_enabled_builtins(): "finish", "run_command", "schedule", + "start_subagent", "view_file", ] - assert caps.enable_subagents is False def test_allowlist_never_enables_tools_the_harness_ships_off(): caps = _capabilities(allowed_tools=["Bash", "Glob", "Grep", "LS", "AskUserQuestion"]) - assert {t.value for t in caps.enabled_tools} == {"run_command", "finish", "schedule"} - - -def test_allowed_task_keeps_subagents(): - caps = _capabilities(allowed_tools=["Bash", "Task"]) - - assert {t.value for t in caps.enabled_tools} == {"run_command", "start_subagent", "finish", "schedule"} - assert caps.enable_subagents is True + assert {t.value for t in caps.enabled_tools} == {"run_command", "finish", "schedule", "start_subagent"} def test_disallowed_tools_map_to_disabled_builtins(): caps = _capabilities(disallowed_tools=["Task", "WebSearch", "TodoWrite"]) - assert [t.value for t in caps.disabled_tools] == ["search_web", "start_subagent"] - assert caps.enable_subagents is False + assert [t.value for t in caps.disabled_tools] == ["search_web"] def test_disallowed_tools_are_removed_from_the_allowlist(): """The SDK takes an allowlist OR a denylist, so with both the denied tools leave the allowlist.""" - caps = _capabilities(allowed_tools=["Bash", "Read", "Task"], disallowed_tools=["Task"]) + caps = _capabilities(allowed_tools=["Bash", "Read", "WebSearch"], disallowed_tools=["WebSearch"]) + + assert {t.value for t in caps.enabled_tools} == {"run_command", "view_file", "finish", "schedule", "start_subagent"} + + +@pytest.mark.parametrize( + ("cfg", "denied"), + [ + ({}, []), + ({"allowed_tools": ["Bash", "Task"]}, []), + ({"allowed_tools": ["Bash", "Read"]}, ["invoke_subagent"]), + ({"disallowed_tools": ["Task"]}, ["invoke_subagent"]), + ({"allowed_tools": ["Bash", "Task"], "disallowed_tools": ["Task"]}, ["invoke_subagent"]), + ], +) +async def test_subagent_calls_are_denied_unless_task_is_allowed(monkeypatch, tmp_path, cfg, denied): + """`start_subagent` stays on so the harness keeps its default system prompt, whose only + "you will be notified, do not poll" guidance sits in its subagents section. A subagent the + tool lists withhold is denied at the call that runs it instead.""" + configs: list[Any] = [] + + class _FakeSdkAgent: + def __init__(self, cfg): + configs.append(cfg) + + async def __aenter__(self): + return self + + async def __aexit__(self, *exc): + return False + + _install_fake_sdk(monkeypatch, _FakeSdkAgent) + + await _agent(**cfg).start(str(tmp_path)) - assert {t.value for t in caps.enabled_tools} == {"run_command", "view_file", "finish", "schedule"} - assert caps.enable_subagents is False + assert [p.tool for p in configs[0].policies if p.kind == "deny"] == denied async def test_start_passes_tool_capabilities_to_sdk_config(monkeypatch, tmp_path): @@ -1622,6 +1647,7 @@ async def __aexit__(self, *exc): "view_file", "finish", "schedule", + "start_subagent", } @@ -1634,7 +1660,7 @@ def test_installed_sdk_accepts_the_tool_capabilities(): assert mapped <= {t.value for t in types.BuiltinTools} caps = _agent(allowed_tools=["Bash", "Read", "Write", "Edit", "Glob", "Grep", "Skill"])._tool_capabilities(types) assert isinstance(caps, types.CapabilitiesConfig) - assert types.BuiltinTools.START_SUBAGENT not in caps.enabled_tools + assert types.BuiltinTools.START_SUBAGENT in caps.enabled_tools assert types.BuiltinTools.FIND_FILE not in caps.enabled_tools From fe532f61348da8463ab6cd4c7fc0a6a4c2d656fd Mon Sep 17 00:00:00 2001 From: Bai Li Date: Sat, 3 Oct 2026 18:01:15 -0700 Subject: [PATCH 5/5] docs(antigravity): state what keeping start_subagent on does and does not fix Co-Authored-By: Claude Opus 5.5 --- .claude/notes/agents.md | 7 ++++--- 1 file changed, 4 insertions(+), 3 deletions(-) diff --git a/.claude/notes/agents.md b/.claude/notes/agents.md index 3b963508..c1c67c28 100644 --- a/.claude/notes/agents.md +++ b/.claude/notes/agents.md @@ -483,9 +483,10 @@ one search per turn. `start_subagent` also stays on under any allowlist, because the harness builds its system prompt from the toolset: with `start_subagent` off it drops the whole subagents section (4,801 → 2,783 characters), and with it the prompt's only "you do NOT need to poll ... you -will be notified" guidance. Without that, Gemini polls a backgrounded command with status -checks and short "liveness" timers: rows doing so rose from 1% (0.12.9) to 8-10%, and -max-turn rows from 2 to 12 of 148. The tool descriptions are identical either way, and +will be notified" guidance. With it on, the system prompt is byte-identical to 0.12.9's. +The section alone does not stop Gemini polling a long job with status checks: in a 48-row +SkillsBench A/B it made 13.4 checks per row with the section and 13.7 without (8.5 on +0.12.9). The tool descriptions are identical either way, and `enable_subagents` alone does not restore the section. Subagents the tool lists withhold (no `Task`) are denied by policy at `invoke_subagent` instead, which covers both the built-in `research` subagent and one the model defines with `define_subagent`; both get