Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
12 changes: 6 additions & 6 deletions benchmarks/agent/test_agent_bench.py
Original file line number Diff line number Diff line change
Expand Up @@ -694,18 +694,18 @@ def test_the_canary_flags_a_network_tool_the_blockers_do_not_shadow(self):

def test_a_withheld_tool_fetched_through_a_package_manager_is_detected_in_native_transcripts(self):
lines = [
{"source": "agent", "tool_calls": [{"function_name": "exec", "arguments": {"command": "npm init -y && npm install --save-dev @wrightkit/wright overpy"}}]},
{"source": "agent", "tool_calls": [{"function_name": "exec", "arguments": {"command": "npm init -y && npm install --save-dev overpy"}}]},
{"payload": {"type": "function_call", "arguments": json.dumps({"cmd": "pip3 install overpy"})}},
{"type": "toolCall", "arguments": {"command": "npx wright check mode.opy"}},
{"type": "toolCall", "arguments": {"command": "npx overpy check mode.opy"}},
{"source": "user", "message": "skill text mentioning `npm install -g overpy` is not a command"},
{"source": "agent", "tool_calls": [{"function_name": "exec", "arguments": {"command": "wright check mode.ws"}}]},
{"source": "agent", "tool_calls": [{"function_name": "exec", "arguments": {"command": "git clone https://github.com/wrightkit/wright"}}]},
]
path = self.out / "transcript.jsonl"
path.write_text("".join(json.dumps(line) + "\n" for line in lines))
found = bench_trace.contraband_installs(path, "none")
self.assertEqual([tool for tool, _ in found], ["wright", "overpy", "overpy", "wright"])
self.assertEqual([tool for tool, _ in bench_trace.contraband_installs(path, "wright")], ["overpy", "overpy"]) # wright is the condition's own tool
self.assertEqual([tool for tool, _ in bench_trace.contraband_installs(path, "overpy")], ["wright", "wright"])
self.assertEqual([tool for tool, _ in found], ["overpy", "overpy", "overpy", "wright"])
self.assertEqual([tool for tool, _ in bench_trace.contraband_installs(path, "wright")], ["overpy", "overpy", "overpy"]) # wright is the condition's own tool
self.assertEqual([tool for tool, _ in bench_trace.contraband_installs(path, "overpy")], ["wright"])

def test_every_blocked_tool_has_a_matching_fetch_pattern_and_every_call_shape_is_read(self):
fetched = [
Expand Down
2 changes: 1 addition & 1 deletion docs/agent-benchmark.md
Original file line number Diff line number Diff line change
Expand Up @@ -237,7 +237,7 @@ or run through a package manager, a repo/release CLI, or a downloader (including
that stay unshimmed, such as `git clone`, `docker pull`, or `scp`) marks the run `invalid`.
Detection keys on the fetch target — a path token like `docs/overpy-notes.md` is not a
fetch; arbitrary code like `python -c 'urllib...'` is outside its scope. This was added
after a baseline run installed `@wrightkit/wright` and `overpy` from npm. Model
after a baseline run fetched the withheld OverPy tool from npm. Model
account usage, CPU and disk consumption remain shared with the host. Provider
failures returned as exit 75 are listed separately and excluded from outcome
metrics. The harness never edits the task prompt: network `off` and the workspace
Expand Down
Loading