From 8bd0ac82ca1ad47c5e80771c8fe4301669b7fed3 Mon Sep 17 00:00:00 2001 From: Teakowa <27560638+Teakowa@users.noreply.github.com> Date: Sat, 10 Oct 2026 01:19:39 +0800 Subject: [PATCH] chore(bench): remove npm Wright package examples --- benchmarks/agent/test_agent_bench.py | 12 ++++++------ docs/agent-benchmark.md | 2 +- 2 files changed, 7 insertions(+), 7 deletions(-) diff --git a/benchmarks/agent/test_agent_bench.py b/benchmarks/agent/test_agent_bench.py index 7aefae84..29d61b05 100644 --- a/benchmarks/agent/test_agent_bench.py +++ b/benchmarks/agent/test_agent_bench.py @@ -694,18 +694,18 @@ def test_the_canary_flags_a_network_tool_the_blockers_do_not_shadow(self): def test_a_withheld_tool_fetched_through_a_package_manager_is_detected_in_native_transcripts(self): lines = [ - {"source": "agent", "tool_calls": [{"function_name": "exec", "arguments": {"command": "npm init -y && npm install --save-dev @wrightkit/wright overpy"}}]}, + {"source": "agent", "tool_calls": [{"function_name": "exec", "arguments": {"command": "npm init -y && npm install --save-dev overpy"}}]}, {"payload": {"type": "function_call", "arguments": json.dumps({"cmd": "pip3 install overpy"})}}, - {"type": "toolCall", "arguments": {"command": "npx wright check mode.opy"}}, + {"type": "toolCall", "arguments": {"command": "npx overpy check mode.opy"}}, {"source": "user", "message": "skill text mentioning `npm install -g overpy` is not a command"}, - {"source": "agent", "tool_calls": [{"function_name": "exec", "arguments": {"command": "wright check mode.ws"}}]}, + {"source": "agent", "tool_calls": [{"function_name": "exec", "arguments": {"command": "git clone https://github.com/wrightkit/wright"}}]}, ] path = self.out / "transcript.jsonl" path.write_text("".join(json.dumps(line) + "\n" for line in lines)) found = bench_trace.contraband_installs(path, "none") - self.assertEqual([tool for tool, _ in found], ["wright", "overpy", "overpy", "wright"]) - self.assertEqual([tool for tool, _ in bench_trace.contraband_installs(path, "wright")], ["overpy", "overpy"]) # wright is the condition's own tool - self.assertEqual([tool for tool, _ in bench_trace.contraband_installs(path, "overpy")], ["wright", "wright"]) + self.assertEqual([tool for tool, _ in found], ["overpy", "overpy", "overpy", "wright"]) + self.assertEqual([tool for tool, _ in bench_trace.contraband_installs(path, "wright")], ["overpy", "overpy", "overpy"]) # wright is the condition's own tool + self.assertEqual([tool for tool, _ in bench_trace.contraband_installs(path, "overpy")], ["wright"]) def test_every_blocked_tool_has_a_matching_fetch_pattern_and_every_call_shape_is_read(self): fetched = [ diff --git a/docs/agent-benchmark.md b/docs/agent-benchmark.md index 12a12163..07e1d8d9 100644 --- a/docs/agent-benchmark.md +++ b/docs/agent-benchmark.md @@ -237,7 +237,7 @@ or run through a package manager, a repo/release CLI, or a downloader (including that stay unshimmed, such as `git clone`, `docker pull`, or `scp`) marks the run `invalid`. Detection keys on the fetch target — a path token like `docs/overpy-notes.md` is not a fetch; arbitrary code like `python -c 'urllib...'` is outside its scope. This was added -after a baseline run installed `@wrightkit/wright` and `overpy` from npm. Model +after a baseline run fetched the withheld OverPy tool from npm. Model account usage, CPU and disk consumption remain shared with the host. Provider failures returned as exit 75 are listed separately and excluded from outcome metrics. The harness never edits the task prompt: network `off` and the workspace