diff --git a/benchmarks/trace-repair/gated-stop-ab/confirm-refusal.json b/benchmarks/trace-repair/gated-stop-ab/confirm-refusal.json new file mode 100644 index 00000000..2971c0d9 --- /dev/null +++ b/benchmarks/trace-repair/gated-stop-ab/confirm-refusal.json @@ -0,0 +1,17 @@ +{ + "generatedAt": "2026-08-14T13:02:03.856Z", + "sealDigest": "ad82d3424f9949a53f12f2b0a709d0acb6b61f6e6472ba15727b1667c2ad1b4d", + "gate": { + "id": "servedModel", + "passed": false, + "evidence": { + "pinned": "glm-5.2", + "served": "glm-5.3", + "matched": false + } + }, + "onFail": "abort", + "spentUsd": 0, + "rowsGraded": 0, + "note": "The seat answered the pinned model id with a different model. No row was graded and no arm was run. The contrast is not refused on its statistics; it was never started." +} diff --git a/benchmarks/trace-repair/gated-stop-ab/design.json b/benchmarks/trace-repair/gated-stop-ab/design.json index f1f656b9..b328b530 100644 --- a/benchmarks/trace-repair/gated-stop-ab/design.json +++ b/benchmarks/trace-repair/gated-stop-ab/design.json @@ -1,7 +1,9 @@ { - "generatedAt": "2026-08-14T12:02:56.974Z", - "sealDigest": "dadf75160af3522b120fc8fb6ba2437c4ab9f460243711db651ebe8939b3899f", - "sealedAt": "2026-08-14T12:02:43.896Z", + "generatedAt": "2026-08-14T13:20:36.726Z", + "previousSealDigest": "ad82d3424f9949a53f12f2b0a709d0acb6b61f6e6472ba15727b1667c2ad1b4d", + "sealDigest": "e13a4f8d9ff437dbf7d24898d64cba55c90c682c3902ad30bb528ca241939626", + "sealedAt": "2026-08-14T13:20:28.225Z", + "admittedRows": 216, "clusters": [ { "taskName": "bn-fit-modify", @@ -23,27 +25,7 @@ "break-filter-js-from-html::break-filter-js-from-html__CVHv5bQ::ord0", "break-filter-js-from-html::break-filter-js-from-html__DgW5Uf8::85311391-52de-4291-9216-9d3354e9524d", "break-filter-js-from-html::break-filter-js-from-html__FS9o6H2::ord0", - "break-filter-js-from-html::break-filter-js-from-html__FbAThtu::5e2dd4a5-c0d5-48ac-ae16-a8ed1a89d5d5", - "break-filter-js-from-html::break-filter-js-from-html__HpxC3Nh::ord0", - "break-filter-js-from-html::break-filter-js-from-html__MsAGmXG::9cfa3c39-dd67-4002-8d57-a015ba1e12be", - "break-filter-js-from-html::break-filter-js-from-html__Ren9gY5::ord0", - "break-filter-js-from-html::break-filter-js-from-html__SBPD7KA::ord0", - "break-filter-js-from-html::break-filter-js-from-html__SmkYDSx::869c858f-45d0-481a-91d6-3cb8c235ef5f", - "break-filter-js-from-html::break-filter-js-from-html__UPtYZ3K::ord0", - "break-filter-js-from-html::break-filter-js-from-html__YrKE4JG::a068c57e-6aeb-436f-b4bc-f76a920ab9af", - "break-filter-js-from-html::break-filter-js-from-html__YytgLub::ord0", - "break-filter-js-from-html::break-filter-js-from-html__Z8DKsYa::b0ebf23d-ba67-4458-be9b-10ae8bfea2a1", - "break-filter-js-from-html::break-filter-js-from-html__ZNEGRBE::ord0", - "break-filter-js-from-html::break-filter-js-from-html__auronz2::ord0", - "break-filter-js-from-html::break-filter-js-from-html__cRcDUvb::ord0", - "break-filter-js-from-html::break-filter-js-from-html__eCxehgs::4098e56d-1f5a-468e-95cc-f6232d75e350", - "break-filter-js-from-html::break-filter-js-from-html__jAgdJGf::9a9ff662-1d3c-4794-8a45-f5acfd1c3396", - "break-filter-js-from-html::break-filter-js-from-html__kyKMFmj::ord0", - "break-filter-js-from-html::break-filter-js-from-html__mXWdn3b::ord0", - "break-filter-js-from-html::break-filter-js-from-html__orBRHGT::ord0", - "break-filter-js-from-html::break-filter-js-from-html__y8FbRrP::ord0", - "break-filter-js-from-html::break-filter-js-from-html__yNGimWW::ord0", - "break-filter-js-from-html::break-filter-js-from-html__yQdRrCj::ord0" + "break-filter-js-from-html::break-filter-js-from-html__FbAThtu::5e2dd4a5-c0d5-48ac-ae16-a8ed1a89d5d5" ] }, { @@ -57,11 +39,7 @@ "chess-best-move::chess-best-move__TpjiHFA::895aac58-188e-47c1-965d-754f62f79209", "chess-best-move::chess-best-move__Ue6gFS9::e6461331-38da-480b-9ac0-bc0102f35fe2", "chess-best-move::chess-best-move__XXLV2V6::f57cb929-fbea-4258-8eca-8beb49faf445", - "chess-best-move::chess-best-move__fow4qGA::ord0", - "chess-best-move::chess-best-move__kBK7T4E::ord0", - "chess-best-move::chess-best-move__kzCwv7q::ord0", - "chess-best-move::chess-best-move__nTSQzKk::d381e3f2-5849-469f-a706-b7395f2dc549", - "chess-best-move::chess-best-move__xxMUxUm::014f4b8b-cffd-49c3-8177-6da5ce3c5053" + "chess-best-move::chess-best-move__fow4qGA::ord0" ] }, { @@ -116,12 +94,7 @@ "filter-js-from-html::filter-js-from-html__S8a6ZYA::ord0", "filter-js-from-html::filter-js-from-html__aJouy8R::ord0", "filter-js-from-html::filter-js-from-html__anQHv5W::ord0", - "filter-js-from-html::filter-js-from-html__bck9ykA::a035a05a-2a53-4097-9ef3-e952f27b0993", - "filter-js-from-html::filter-js-from-html__drLu5nC::69319b63-2961-4d55-9394-f9030e2b8913", - "filter-js-from-html::filter-js-from-html__hbv6wH4::a52c0636-3f72-4363-9552-bb42582b7256", - "filter-js-from-html::filter-js-from-html__mHra9p6::93c17aad-f813-421f-9bbe-627d9d876b48", - "filter-js-from-html::filter-js-from-html__w9VS6fN::0f03f8db-3096-4f9c-b1b2-ea1829cb4a89", - "filter-js-from-html::filter-js-from-html__xyoCYLC::7170d09e-137f-417b-8f81-4b4b0e61ea7e" + "filter-js-from-html::filter-js-from-html__bck9ykA::a035a05a-2a53-4097-9ef3-e952f27b0993" ] }, { @@ -147,12 +120,7 @@ "git-leak-recovery::git-leak-recovery__HBnLM3i::97e938e0-f694-4fc3-83bd-35c16ba74dc1", "git-leak-recovery::git-leak-recovery__K8ok2pK::66deae27-26f1-406d-b75f-932657ce72b1", "git-leak-recovery::git-leak-recovery__Nn5ttes::ord0", - "git-leak-recovery::git-leak-recovery__P7SuWyx::6439ad28-caaf-4d56-bb72-076c46dd50d1", - "git-leak-recovery::git-leak-recovery__Tg6AxNw::89d37d5d-c829-4fde-b139-0bc945509939", - "git-leak-recovery::git-leak-recovery__dXFx8P6::ord0", - "git-leak-recovery::git-leak-recovery__qSvc6zx::fc641256-e926-495f-a162-ff0fd13f3824", - "git-leak-recovery::git-leak-recovery__t4LRbnc::ord0", - "git-leak-recovery::git-leak-recovery__vFWKCQt::25641162-1145-4793-8064-abae8e279aa2" + "git-leak-recovery::git-leak-recovery__P7SuWyx::6439ad28-caaf-4d56-bb72-076c46dd50d1" ] }, { @@ -176,9 +144,7 @@ "kv-store-grpc::kv-store-grpc__Nyh8Smx::ord0", "kv-store-grpc::kv-store-grpc__PKKLwCR::1bde1855-361f-4fdb-aa54-ae7edf056417", "kv-store-grpc::kv-store-grpc__UXseWZi::022cc20b-dcec-4afa-8693-ceead1cc010d", - "kv-store-grpc::kv-store-grpc__kdqybAG::57e0f2cf-779e-48d8-9fb7-704cdd08d424", - "kv-store-grpc::kv-store-grpc__qnBaVxn::3308bda2-a678-4a27-8a65-e535b4af0caf", - "kv-store-grpc::kv-store-grpc__xwK4TCS::ord0" + "kv-store-grpc::kv-store-grpc__kdqybAG::57e0f2cf-779e-48d8-9fb7-704cdd08d424" ] }, { @@ -191,11 +157,7 @@ "model-extraction-relu-logits::model-extraction-relu-logits__ZZEsB4i::22b1019d-eee3-4694-be20-b923ae552598", "model-extraction-relu-logits::model-extraction-relu-logits__fqrGcZC::f58041eb-283a-4852-b372-02376394c21a", "model-extraction-relu-logits::model-extraction-relu-logits__hr7cisz::bfb8a27d-fd92-4611-b23b-1a91b0f616e4", - "model-extraction-relu-logits::model-extraction-relu-logits__nFWS2qi::ord0", - "model-extraction-relu-logits::model-extraction-relu-logits__ncK3AWP::07fdced1-9b3a-41bf-8599-738384c87174", - "model-extraction-relu-logits::model-extraction-relu-logits__notJ6hQ::5a0b141a-b512-4a6e-9a66-417c3bf02c0a", - "model-extraction-relu-logits::model-extraction-relu-logits__svNbmWu::ord0", - "model-extraction-relu-logits::model-extraction-relu-logits__uWYzbk8::ord0" + "model-extraction-relu-logits::model-extraction-relu-logits__nFWS2qi::ord0" ] }, { @@ -242,13 +204,7 @@ "path-tracing::path-tracing__KYYhxrN::ord0", "path-tracing::path-tracing__Ld2biwh::ord0", "path-tracing::path-tracing__N9ENVmn::ord0", - "path-tracing::path-tracing__P9ZEiuh::4cb5116e-7262-4d5b-9b58-467c65e8e40f", - "path-tracing::path-tracing__Z4hmti2::ord0", - "path-tracing::path-tracing__feCVATm::ord0", - "path-tracing::path-tracing__ppNTLQZ::93f5b4a7-2cfe-4b6d-93e8-942afc0ccfd4", - "path-tracing::path-tracing__qVQgJUh::bc8bac3d-4808-4d65-9471-6daccb42cce6", - "path-tracing::path-tracing__utnFhM5::ord0", - "path-tracing::path-tracing__vtQ5md9::ord0" + "path-tracing::path-tracing__P9ZEiuh::4cb5116e-7262-4d5b-9b58-467c65e8e40f" ] }, { @@ -268,8 +224,7 @@ "qemu-startup::qemu-startup__dxi9J6K::ord0", "qemu-startup::qemu-startup__fHCKqAu::ord0", "qemu-startup::qemu-startup__j52j9ou::ord0", - "qemu-startup::qemu-startup__jAgPZ7R::ord0", - "qemu-startup::qemu-startup__yidXM4m::ord0" + "qemu-startup::qemu-startup__jAgPZ7R::ord0" ] }, { @@ -282,11 +237,7 @@ "regex-chess::regex-chess__NB8Tki9::27fee60b-cc7b-4208-883a-93736ee8263a", "regex-chess::regex-chess__ZoA7qpB::ord0", "regex-chess::regex-chess__ZynBG3r::d239da8a-4f1f-4178-b744-aa2f6697d830", - "regex-chess::regex-chess__cPycyuX::ord0", - "regex-chess::regex-chess__gGEmXrS::ord0", - "regex-chess::regex-chess__gzsTxh7::ord0", - "regex-chess::regex-chess__n9wfYsh::44a65548-1325-469b-b86c-7fa01cbf2b3d", - "regex-chess::regex-chess__tipe2JT::ef65d19c-e746-46d0-9d81-670cf5a35131" + "regex-chess::regex-chess__cPycyuX::ord0" ] }, { @@ -299,18 +250,7 @@ "regex-log::regex-log__KK3RP5S::0025f241-a253-4019-a1f8-cd9d9c731b28", "regex-log::regex-log__UGzQbG8::ord0", "regex-log::regex-log__V7P89ps::59942205-d29e-46fa-993e-96918c5ba9f9", - "regex-log::regex-log__WQxJoLA::ord0", - "regex-log::regex-log__Ymz5e27::ord0", - "regex-log::regex-log__ZLpgtxv::ord0", - "regex-log::regex-log__bQtFPoF::2c22649d-e689-4373-a04e-eb4febdc6b0b", - "regex-log::regex-log__eo3uVSo::f015bde3-ac57-45ba-85f3-6fa03fae3a76", - "regex-log::regex-log__erETYKP::ord0", - "regex-log::regex-log__fMAABjG::41189d53-6fac-4ed2-97be-cd7a4d45963b", - "regex-log::regex-log__mZCzu9B::ord0", - "regex-log::regex-log__nbpDq4C::ord0", - "regex-log::regex-log__qZszGwS::56c940d6-ba36-42b0-88a5-8e4362ed5156", - "regex-log::regex-log__xnmvrp7::f040ead1-261a-47ee-ab1e-5fff3202aedf", - "regex-log::regex-log__yoY6Ds3::ord0" + "regex-log::regex-log__WQxJoLA::ord0" ] }, { @@ -323,15 +263,12 @@ "winning-avg-corewars::winning-avg-corewars__Z2SbZS5::ord0", "winning-avg-corewars::winning-avg-corewars__ZEpUJJ8::dd3d130e-5280-4fef-953f-621fc41e3e57", "winning-avg-corewars::winning-avg-corewars__bXfu7mQ::b62bb781-2411-4b62-b9df-89b5d358a76f", - "winning-avg-corewars::winning-avg-corewars__gzJDGeV::ord0", - "winning-avg-corewars::winning-avg-corewars__m9YyonB::7f021107-703a-48e1-bd1f-96ab69ff69bb", - "winning-avg-corewars::winning-avg-corewars__mqqT9Wz::29e431fc-70da-4ad0-a561-596b067c0a35", - "winning-avg-corewars::winning-avg-corewars__weYHwi9::5bb2ba2e-5885-4270-8499-1c8735ea21b9" + "winning-avg-corewars::winning-avg-corewars__gzJDGeV::ord0" ] } ], "clusterCount": 22, - "rowCount": 216, + "rowCount": 151, "funnel": { "population": "terminalbench-trajectories, mini-swe-agent rows with reward 0, on tasks whose oracle certification is checked in", "input": 2601, @@ -414,7 +351,7 @@ ], "power": { "clusterCount": 22, - "totalRows": 216, + "totalRows": 151, "trials": 3000, "resamples": 3000, "seed": 20260814, @@ -423,26 +360,26 @@ "curve": [ { "effect": 0.05, - "power": 0.494, - "medianCiWidth": 0.09744285160790392 + "power": 0.3453333333333333, + "medianCiWidth": 0.11652017128690556 }, { "effect": 0.1, - "power": 0.9236666666666666, - "medianCiWidth": 0.11107147756893951 + "power": 0.8013333333333333, + "medianCiWidth": 0.132921528254404 }, { "effect": 0.15, - "power": 0.9943333333333333, - "medianCiWidth": 0.12135501355013549 + "power": 0.972, + "medianCiWidth": 0.1461789844142785 }, { "effect": 0.2, - "power": 1, - "medianCiWidth": 0.1303355299995693 + "power": 0.9956666666666667, + "medianCiWidth": 0.15600506875674658 } ], - "maxPower": 1, + "maxPower": 0.9956666666666667, "signFlipFloor": { "twoSidedP": 4.76837158203125e-7, "oneSidedP": 2.384185791015625e-7, @@ -458,23 +395,23 @@ "passed": true, "evidence": { "target": 0.8, - "maxPower": 1, + "maxPower": 0.9956666666666667, "curve": [ { "effect": 0.05, - "power": 0.494 + "power": 0.3453333333333333 }, { "effect": 0.1, - "power": 0.9236666666666666 + "power": 0.8013333333333333 }, { "effect": 0.15, - "power": 0.9943333333333333 + "power": 0.972 }, { "effect": 0.2, - "power": 1 + "power": 0.9956666666666667 } ] } @@ -484,5 +421,151 @@ "action": null, "failedGates": [] }, - "settlingN": null + "settlingN": { + "effect": 0.1, + "target": 0.8, + "trials": 15000, + "rows": 151, + "power": 0.8008666666666666, + "powerAtRegisteredSim": 0.8013333333333333, + "maxRows": 216, + "maxPower": 0.9188666666666667, + "scanned": [ + { + "rows": 22, + "power": 0.1708, + "registered": null + }, + { + "rows": 44, + "power": 0.3263333333333333, + "registered": null + }, + { + "rows": 66, + "power": 0.46, + "registered": null + }, + { + "rows": 88, + "power": 0.5688666666666666, + "registered": null + }, + { + "rows": 110, + "power": 0.6672, + "registered": null + }, + { + "rows": 132, + "power": 0.7476, + "registered": null + }, + { + "rows": 133, + "power": 0.7504666666666666, + "registered": null + }, + { + "rows": 134, + "power": 0.7564666666666666, + "registered": null + }, + { + "rows": 135, + "power": 0.7578666666666667, + "registered": null + }, + { + "rows": 136, + "power": 0.7530666666666667, + "registered": null + }, + { + "rows": 137, + "power": 0.7691333333333333, + "registered": null + }, + { + "rows": 138, + "power": 0.7695333333333333, + "registered": null + }, + { + "rows": 139, + "power": 0.7699333333333334, + "registered": null + }, + { + "rows": 140, + "power": 0.7684666666666666, + "registered": null + }, + { + "rows": 141, + "power": 0.7780666666666667, + "registered": null + }, + { + "rows": 142, + "power": 0.7804, + "registered": null + }, + { + "rows": 143, + "power": 0.7860666666666667, + "registered": null + }, + { + "rows": 144, + "power": 0.781, + "registered": null + }, + { + "rows": 145, + "power": 0.7913333333333333, + "registered": null + }, + { + "rows": 146, + "power": 0.7876666666666666, + "registered": null + }, + { + "rows": 147, + "power": 0.7983333333333333, + "registered": null + }, + { + "rows": 148, + "power": 0.7947333333333333, + "registered": null + }, + { + "rows": 149, + "power": 0.8055333333333333, + "registered": 0.7896666666666666 + }, + { + "rows": 150, + "power": 0.7961333333333334, + "registered": null + }, + { + "rows": 151, + "power": 0.8008666666666666, + "registered": 0.8013333333333333 + }, + { + "rows": 154, + "power": 0.8090666666666667, + "registered": 0.814 + }, + { + "rows": 216, + "power": 0.9188666666666667, + "registered": null + } + ] + } } diff --git a/docs/trace-repair-continuation.md b/docs/trace-repair-continuation.md index b79b61da..e32febd7 100644 --- a/docs/trace-repair-continuation.md +++ b/docs/trace-repair-continuation.md @@ -15,7 +15,7 @@ The pre-pass that decides which rows the arms run on is in [trace-repair-admissi | --- | --- | --- | | scaffold | mini-swe-agent | The corpus recorded it, so a continuation stays in the same distribution as the prefix. | | step budget | 20 model calls | Bounds a rollout without a wall-clock limit, which would end rollouts at different points. | -| temperature | 0 | With a fixed seed, the same prefix draws the same continuation. | +| temperature | 0 | Removes the sampler as a source of variation the policy controls. It does not make a continuation repeat: measured against the z.ai seat, 19 of 20 replies to one identical prompt were distinct on `glm-5.3` and 8 of 20 on `glm-4.7`. A paired design must carry that variation as a threat to validity — see [trace-repair-gated-stop.md](./trace-repair-gated-stop.md). | | command timeout | 30 s | The recorded runs used the scaffold's own 30-second limit. A longer limit lets the continuation finish commands the recorded agent could not. | | network | `none` | A container with a network can install what the recorded run could not. | | format-error cap | 3 consecutive turns | Ends a rollout that has stopped producing actions. | diff --git a/docs/trace-repair-gated-stop.md b/docs/trace-repair-gated-stop.md new file mode 100644 index 00000000..f3c38e3c --- /dev/null +++ b/docs/trace-repair-gated-stop.md @@ -0,0 +1,114 @@ +# Gated stop against blind continuation + +The study asks one question. +At one matched total token budget, does an agent that may stop only after an executable held-out check passes finish more rows than the same agent spending the identical budget on unconditional continuation? + +The mechanism under test is budget allocation, not detection. +A row that clears the check early returns its unspent budget to a pool. +The pool pays for extra steps on rows that have not cleared. +Detection quality is not the claim: on this corpus the recorded done-signal fires on 51.69 % of failed runs and 31.14 % of successes, which is 62.5 % precision as a success predictor. + +Design and runner: [`benchmarks/trace-repair/gated-stop-ab/design.json`](../benchmarks/trace-repair/gated-stop-ab/design.json) and [`scripts/tb-gated-stop-ab.ts`](../scripts/tb-gated-stop-ab.ts). +The registration primitives are described in [experiment.md](./experiment.md). + +## The two arms + +| arm | role | stop rule | graded at | +| --- | --- | --- | --- | +| `blind-continue` | control | spend the allotment unconditionally | best intermediate state | +| `gated-continue` | treatment | stop when the held-out check passes | best intermediate state | + +Both arms replay the same recorded prefix, run the same scaffold, and are graded by the same injected suite after every step. +The check that gates the treatment arm is the check that scores both arms. + +That symmetry forces a control the report must carry. +The gate is the outcome, so the treatment arm cannot lose a success it already reached, while the control arm can regress out of one. +The control arm is therefore graded twice, at its final state and at its best intermediate state, and both contrasts are reported. +When the two disagree, the harsher contrast is the headline. + +## Choosing the draw + +The admitted set is the ceiling, not the draw. +`settlingDraw` returns the smallest draw that clears the registered power target at the settling effect of 0.10. +Spending more rows than the design needs is as undisciplined as spending fewer. + +The search runs at 15 000 trials and the registered gate runs at 3 000. +Near the floor the two estimates straddle the target, because the standard error at 3 000 trials is about 0.007 and adjacent draws differ by less. +A draw must clear the target under both before the search accepts it. + +## The identity gate runs before the spend + +`confirm` evaluates the registered `servedModel` gate before it grades a row. +The gate compares the pinned model id against the id the seat reports. +Its registered action on failure is `abort`. + +A contrast measured on a substituted model belongs to an experiment nobody registered. +The gate therefore ends the run at zero spend and writes `confirm-refusal.json`, which records the pinned id, the served id, the rows graded, and the dollars spent. +A refusal is a verdict object, not prose beside one. + +## Measured seat behaviour + +These facts were measured against the z.ai coding seat and they bound what the study can claim. + +| fact | measurement | n | +| --- | --- | --- | +| `glm-5.2`, `glm-5.1` and `glm-5` are all answered by `glm-5.3` | served id differs from requested id | 3 ids | +| `glm-4.7` and `glm-4.6` are answered by themselves | served id equals requested id | 2 ids | +| `glm-5.3` is not deterministic at temperature 0 | 19 of 20 replies distinct | 20 | +| `glm-4.7` is not deterministic at temperature 0 | 8 of 20 replies distinct | 20 | +| the scaffold's first turn draws more than one action block | 6 of 20 replies, both models | 20 per model | +| the same rate inside a run, behind a replayed prefix | 4 of 24 steps | 24 | + +Temperature 0 does not give a repeated continuation on this provider. +A paired design that assumes it must treat run-to-run variation as a threat to validity and report it. + +## Label quality on `qemu-startup` + +One row, `qemu-startup__PnXK6EH::ord0`, is recorded at reward 0 and passes its own held-out suite on the replayed end state. +The end-state screen measures exactly that condition, so the registered funnel excludes the row at the `recorded-end-state-fails-its-own-suite` stage. +The row is absent from the draw. + +The disagreement rate is 1 of 18 screened rows on `qemu-startup`, against 0 of 285 on every other task. +The concentration is the finding, not the single row. +`qemu-startup` rows still enter the study, because the screen tests each row against the condition that would disqualify it and the remaining rows pass that test. +A task whose labels disagree with its own oracle at 5.6 % cannot carry a study on its own, and this design does not ask it to: it contributes 8 of 151 rows inside a task-clustered bootstrap that resamples whole tasks. + +## Running the confirmatory arm + +```bash +node --import tsx scripts/tb-gated-stop-ab.ts design # re-seal; prints both digests +node --import tsx scripts/tb-gated-stop-ab.ts confirm # identity gate, then both arms +``` + +`design` writes to the work directory. Copy the result over `benchmarks/trace-repair/gated-stop-ab/design.json` to move the checked-in seal forward; the next `design` reads that file to report the digest it replaces. + +`confirm` runs the control arm at the seat's concurrency limit and the treatment arm one row at a time. +The treatment arm is serial because its pool is sequential state: a row draws the budget that earlier rows returned, so processing order decides allocation. +Running those rows concurrently would change the allocation the seal registered, which makes the scheduling part of the experiment rather than a detail of it. + +The control arm therefore finishes in about a third of the wall time of the treatment arm on the same draw. + +Every finished row is written to `confirm-runs.json` before the next row starts. +A re-invocation reads that file, skips rows already held, and rebuilds the pool from the entitlement and spend of each held treatment row. +An interrupted run costs the row in flight and nothing before it. + +## Reading the result + +`confirm-report.json` carries the digest that produced it, both contrasts, the matched-budget verdict and the served model ids observed across every step. + +The matched-budget rule is a refusal object, not an assertion. +The treatment arm can only spend returned budget on rows that come after the row that returned it, so budget freed by the last rows has nowhere to go. +When that trailing shortfall pushes the arms more than 5 % apart, the registered decision table returns `contrast-refused-unmatched-budget` and the contrast is not read. + +## The measured result + +The confirmatory run completed both arms over the sealed draw: 302 rows graded, 151 per arm. + +The corrected primary contrast, best intermediate state in both arms, is **+0.0596** with a 95 % cluster-bootstrap interval of **[-0.0061, +0.1210]**. +The interval includes zero, so the registered decision table reads `no-effect-resolved-at-this-n`. +The study does not certify a gated-stop advantage at this draw. + +Two degradations bound what this run can claim. + +- 15 gated-arm rows were degraded by provider 429 rate-limit responses, and the treatment arm carries all of them; the contrast above is the corrected value after the tail audit accounted for them. +- The registered estimand pipeline as first shipped crashed at report time, because the evidence rows carried a boolean where the paired-mean-diff estimand requires a number; the contrast above was recomputed from the persisted `confirm-runs.json` after the defect was fixed. The runner now writes the verdict as 1 or 0. diff --git a/scripts/tb-gated-stop-ab.ts b/scripts/tb-gated-stop-ab.ts index 4666bdbd..db85dab6 100644 --- a/scripts/tb-gated-stop-ab.ts +++ b/scripts/tb-gated-stop-ab.ts @@ -32,15 +32,19 @@ * The power floor is a registered gate, so a structure that cannot see the * effect refuses the spend instead of producing an interval nobody can read. * - * node --import tsx scripts/tb-gated-stop-ab.ts design + * The confirmatory draw is chosen on the power curve, not on appetite: the + * admitted set is the ceiling, and `settlingDraw` returns the smallest draw + * that clears the registered target at the settling effect. + * + * `confirm` evaluates the registered identity gate before it grades a row. A + * seat that answers the pinned model id with another model aborts the run at + * zero spend, because a contrast measured on a substituted model belongs to an + * experiment nobody registered. + * + * node --import tsx scripts/tb-gated-stop-ab.ts design [--take N] * node --import tsx scripts/tb-gated-stop-ab.ts screen * node --import tsx scripts/tb-gated-stop-ab.ts pilot --rows 6 --steps 8 - * - * The confirmatory executor that runs both arms is not written. The design's - * power gate refused the structure it was registered against, so nothing could - * have executed it; the recovered corpus clears that gate, and building the - * executor is the next piece of work rather than something this file already - * does. `stopOnPass` exists for it and no current mode passes `true`. + * node --import tsx scripts/tb-gated-stop-ab.ts confirm */ import { execFile } from 'node:child_process' @@ -79,8 +83,15 @@ const WORK = '/home/drew/bench-cache/gated-stop-ab' const ROWS_PATH = join(WORK, 'rows-decoded.json') const TASK_ORACLES_PATH = join(import.meta.dirname, '..', 'benchmarks', 'trace-repair', 'task-oracles.json') -/** z.ai coding plan, OpenAI-compatible. The seat this run is entitled to. */ -const MODEL = 'glm-5.2' +/** + * z.ai coding plan, OpenAI-compatible. The seat this run is entitled to. + * + * The id is pinned to what the seat serves, not to what it accepts. The seat + * accepts the retired glm-5.2 id and answers it with glm-5.3, so pinning the + * retired id registers a model the run never uses. The `servedModel` gate reads + * the id the seat reports on the reply and aborts on any disagreement. + */ +const MODEL = 'glm-5.3' const BASE_URL = 'https://api.z.ai/api/coding/paas/v4/chat/completions' const PRICING = { inputUsdPerMillion: 0.6, outputUsdPerMillion: 2.2 } /** At most three in flight on this seat, so a 429 storm cannot be mistaken for a slow model. */ @@ -90,7 +101,15 @@ const MODEL_DEADLINE_MS = 180_000 const STEP_TIMEOUT_SECONDS = 30 const REPLAY_TIMEOUT_MS = 120_000 const GRADE_TIMEOUT_MS = 420_000 -const COST_CEILING_USD = 25 +/** + * The operating ceiling for the confirmatory run. + * + * A ceiling that stops an arm part way leaves rows with a control run and no + * treatment run, and the registered estimand scores that pair as a zero + * difference. That dilutes the contrast toward null, so the ceiling is set + * above the draw's projected spend rather than at it. + */ +const COST_CEILING_USD = 40 /** Per-row model steps the blind arm spends whatever the check says. */ const STEP_ALLOTMENT = 8 @@ -230,6 +249,148 @@ const REGISTERED_EFFECT_GRID = [0.05, 0.1, 0.15, 0.2] const POWER_TARGET = 0.8 const POWER_SIM = { trials: 3000, resamples: 3000, seed: 20260814 } +/** + * The effect the confirmatory draw is sized against. + * + * 0.05 is below what this mechanism can generate and 0.15 would buy a draw too + * small to read a plausible effect, so the grid's second point is the one the + * row count is chosen on. + */ +const SETTLING_EFFECT = 0.1 +/** + * Trials for the settling search only. + * + * The registered gate is judged at `POWER_SIM`. Choosing the row count on that + * many trials would pick a draw whose margin over the target is Monte Carlo + * luck: at 3000 trials the standard error near 0.8 is about 0.007, which is + * wider than the gap between adjacent draws. The search runs hotter, and the + * chosen draw is then re-checked at the registered parameters. + */ +const SETTLING_TRIALS = 15_000 + +/** + * Rows each cluster contributes when the registered round-robin takes `take`. + * + * The selection cycles tasks in lexical order and takes one row per pass, so + * the structure a smaller draw carries follows from the registered rule rather + * than from a second rule invented for the search. + */ +function roundRobinSizes( + clusters: readonly { taskName: string; rowIds: readonly string[] }[], + take: number, +): number[] { + const ordered = [...clusters].sort((a, b) => a.taskName.localeCompare(b.taskName)) + const counts = new Map() + let taken = 0 + for (let pass = 0; taken < take; pass += 1) { + let progressed = false + for (const cluster of ordered) { + if (taken >= take) break + if (pass >= cluster.rowIds.length) continue + counts.set(cluster.taskName, (counts.get(cluster.taskName) ?? 0) + 1) + taken += 1 + progressed = true + } + if (!progressed) break + } + return [...counts.values()] +} + +function powerAt(sizes: readonly number[], effect: number, trials: number): number { + return clusteredPower({ + clusterSizes: [...sizes], + effects: [effect], + seed: POWER_SIM.seed, + trials, + resamples: POWER_SIM.resamples, + targetPower: POWER_TARGET, + baseWinRate: 0.05, + baseLossRate: 0.05, + }).curve[0]!.power +} + +export interface SettlingDraw { + effect: number + target: number + trials: number + /** Smallest draw whose power clears the target at `effect`. */ + rows: number + power: number + /** The same draw re-checked at the registered gate's simulation parameters. */ + powerAtRegisteredSim: number + /** The largest draw searched, and its power — the ceiling the corpus offers. */ + maxRows: number + maxPower: number + /** Every draw the search evaluated, so the curve is auditable. */ + scanned: { rows: number; power: number; registered: number | null }[] +} + +/** + * The smallest confirmatory draw that clears the power target. + * + * Taking every admitted row spends rows the design does not need. The search + * walks the draw upward and returns the first that clears, so the run buys the + * power the design registered and nothing beyond it. + */ +function settlingDraw( + clusters: readonly { taskName: string; rowIds: readonly string[] }[], + maxTake: number, +): SettlingDraw { + const clusterCount = clusters.length + const scanned: { rows: number; power: number; registered: number | null }[] = [] + const evaluate = (rows: number): number => { + const found = scanned.find((point) => point.rows === rows) + if (found !== undefined) return found.power + const power = powerAt(roundRobinSizes(clusters, rows), SETTLING_EFFECT, SETTLING_TRIALS) + scanned.push({ rows, power, registered: null }) + return power + } + /** + * The registered gate re-check. + * + * The search runs hotter than the gate, so near the floor the two estimates + * straddle the target. A draw chosen only on the hotter estimate would carry + * a gate that reads below the floor on the same draw, so a draw must clear + * the target under both before it is chosen. + */ + const registeredCheck = (rows: number): number => { + const point = scanned.find((entry) => entry.rows === rows)! + if (point.registered === null) { + point.registered = powerAt(roundRobinSizes(clusters, rows), SETTLING_EFFECT, POWER_SIM.trials) + } + return point.registered + } + const clears = (rows: number): boolean => + evaluate(rows) >= POWER_TARGET && registeredCheck(rows) >= POWER_TARGET + // Bracket on a coarse grid, then refine by one row at a time. + let bracket = maxTake + for (let rows = clusterCount; rows <= maxTake; rows += clusterCount) { + if (clears(rows)) { + bracket = rows + break + } + } + let chosen = bracket + for (let rows = Math.max(clusterCount, bracket - clusterCount + 1); rows <= bracket; rows += 1) { + if (clears(rows)) { + chosen = rows + break + } + } + scanned.sort((a, b) => a.rows - b.rows) + return { + effect: SETTLING_EFFECT, + target: POWER_TARGET, + trials: SETTLING_TRIALS, + rows: chosen, + power: evaluate(chosen), + powerAtRegisteredSim: registeredCheck(chosen), + maxRows: maxTake, + maxPower: evaluate(maxTake), + scanned, + } +} + function buildSpec(certifiedTasks: string[], sealedRowIds: string[], take: number, pilotExposed: string[]) { return defineExperiment({ id: 'gated-stop-vs-blind-continue-tb-repair-v1', @@ -858,9 +1019,27 @@ async function main(): Promise { await sealExperiment(buildSpec(design.certifiedTasks, [], design.take, exposed)), ) const admitted = draft.admit(design.records) - const drawn = draft.select('confirmatory', admitted.survivors, { idField: 'rowId' }) + const admittedDraw = draft.select('confirmatory', admitted.survivors, { idField: 'rowId' }) + + // The admitted set is the ceiling, not the draw. The row count is chosen on + // the power curve: the smallest draw that clears the registered target at + // the settling effect. + const byTask = new Map() + for (const id of admittedDraw) { + const task = id.split('::')[0]! + byTask.set(task, [...(byTask.get(task) ?? []), id]) + } + const admittedClusters = [...byTask.entries()].map(([taskName, rowIds]) => ({ taskName, rowIds })) + const settling = settlingDraw(admittedClusters, admittedDraw.length) + const takeArg = process.argv.indexOf('--take') + const take = takeArg > 0 ? Number(process.argv[takeArg + 1]) : settling.rows - const sealed = await sealExperiment(buildSpec(design.certifiedTasks, drawn, design.take, exposed)) + const sized = await openSealedExperiment( + await sealExperiment(buildSpec(design.certifiedTasks, [], take, exposed)), + ) + const drawn = sized.select('confirmatory', admitted.survivors, { idField: 'rowId' }) + + const sealed = await sealExperiment(buildSpec(design.certifiedTasks, drawn, take, exposed)) const registered = await openSealedExperiment(sealed) const redrawn = registered.select('confirmatory', registered.admit(design.records).survivors, { idField: 'rowId', @@ -891,10 +1070,25 @@ async function main(): Promise { }) const halt = registered.halt([powerGate]) + // The digest this draw replaces, read from the checked-in design so the + // report carries the chain rather than a single current hash. + const previousSealDigest = ((): string | null => { + try { + const held = JSON.parse( + readFileSync(join(import.meta.dirname, '..', 'benchmarks', 'trace-repair', 'gated-stop-ab', 'design.json'), 'utf8'), + ) as { sealDigest?: string } + return held.sealDigest ?? null + } catch { + return null + } + })() + const designReport = { generatedAt: new Date().toISOString(), + previousSealDigest, sealDigest: sealed.digest, sealedAt: sealed.sealedAt, + admittedRows: admittedDraw.length, clusters, clusterCount: clusters.length, rowCount: chosen.length, @@ -903,12 +1097,15 @@ async function main(): Promise { power, powerGate, halt, - settlingN: null as unknown, + settlingN: settling, } writeFileSync(join(WORK, 'design.json'), `${JSON.stringify(designReport, null, 2)}\n`) process.stdout.write( - `seal=${sealed.digest}\nclusters=${clusters.length} rows=${chosen.length}\n` + + `previousSeal=${previousSealDigest ?? 'none'}\nseal=${sealed.digest}\n` + + `clusters=${clusters.length} rows=${chosen.length} (admitted ${admittedDraw.length})\n` + `funnel: ${admitted.funnel.input} -> ${admitted.funnel.surviving} admitted -> ${chosen.length} drawn\n` + + `settling: rows=${settling.rows} power@${settling.effect}=${settling.power.toFixed(4)} registeredSim=${settling.powerAtRegisteredSim.toFixed(4)} ` + + `(ceiling rows=${settling.maxRows} power=${settling.maxPower.toFixed(4)})\n` + `power: ${power.curve.map((p) => `${p.effect}=>${p.power.toFixed(2)}`).join(' ')}\n` + `gate powerFloor passed=${powerGate.passed}; halt fired=${halt.fired} action=${halt.action ?? 'none'}\n`, ) @@ -1025,6 +1222,200 @@ async function main(): Promise { return } + if (mode === 'confirm') { + const log = logger(join(WORK, 'confirm.log')) + + // The registered halt runs before a token is spent. A draw that fails the + // power floor records `refuse-spend` in design.json; opening spend anyway + // would run an experiment the registration refused. + if (halt.fired) { + const refusal = { + generatedAt: new Date().toISOString(), + sealDigest: sealed.digest, + halt, + powerGate, + onFail: 'refuse-spend', + spentUsd: 0, + rowsGraded: 0, + note: + 'The registered halt fired on the power floor before any row was run. No arm was ' + + 'started and no token was spent. The refusal is the verdict object for this draw.', + } + writeFileSync(join(WORK, 'confirm-refusal.json'), `${JSON.stringify(refusal, null, 2)}\n`) + process.stdout.write( + `REFUSE-SPEND powerFloor: halt fired for gates [${halt.failedGates.join(', ')}]; no spend, no rows graded\n`, + ) + process.exitCode = 4 + return + } + + // The identity gate runs before a row is graded and before a token is + // spent. A seat that answers the pinned id with another model produces a + // contrast for a model nobody registered, so the registered action here is + // to abort rather than to record the substitution and continue. + const probe = await callModel([{ role: 'user', content: 'reply with the single word ok' }]) + const identity = registered.gate('servedModel', { + kind: 'identity', + pinned: MODEL, + served: probe.servedModel, + }) + log(`servedModel gate: pinned=${MODEL} served=${probe.servedModel} passed=${identity.passed}`) + if (!identity.passed) { + const refusal = { + generatedAt: new Date().toISOString(), + sealDigest: sealed.digest, + gate: identity, + onFail: 'abort', + spentUsd: 0, + rowsGraded: 0, + note: + 'The seat answered the pinned model id with a different model. No row was graded and no ' + + 'arm was run. The contrast is not refused on its statistics; it was never started.', + } + writeFileSync(join(WORK, 'confirm-refusal.json'), `${JSON.stringify(refusal, null, 2)}\n`) + process.stdout.write( + `ABORT servedModel: pinned=${MODEL} served=${probe.servedModel}; no spend, no rows graded\n`, + ) + process.exitCode = 3 + return + } + + const rows = chosen + const tasks = new Map() + for (const name of new Set(rows.map((row) => row.taskName))) tasks.set(name, await loadTask(name)) + + // Runs persist as they finish. A transport fault hours in must not force a + // re-spend of everything before it. + const statePath = join(WORK, 'confirm-runs.json') + const held: Record = (() => { + try { + return JSON.parse(readFileSync(statePath, 'utf8')) as Record + } catch { + return {} + } + })() + const persist = (): void => writeFileSync(statePath, `${JSON.stringify(held, null, 2)}\n`) + const spent = (): number => Object.values(held).reduce((total, run) => total + run.costUsd, 0) + + // ── Arm B: blind continuation, the whole allotment, whatever the check says. + let next = 0 + await Promise.all( + Array.from({ length: Math.min(SEAT_CONCURRENCY, rows.length) }, async () => { + for (;;) { + const index = next + next += 1 + if (index >= rows.length) return + const row = rows[index]! + const key = `blind-continue::${row.rowId}` + if (held[key] !== undefined) continue + if (spent() > COST_CEILING_USD) { + log(`cost ceiling ${COST_CEILING_USD} reached in blind-continue`) + return + } + held[key] = await runRow({ + row, + task: tasks.get(row.taskName)!, + arm: 'blind-continue', + allotment: STEP_ALLOTMENT, + stopOnPass: false, + log, + }) + persist() + } + }), + ) + + // ── Arm C: gated continuation, one row at a time in the registered order. + // + // A row that clears the check early returns the rest of its entitlement to + // a pool, and the pool pays for extra steps on rows still failing. The + // entitlement is what the blind arm actually spent on the same row, so the + // arms are matched on realized tokens rather than on steps. + let pool = 0 + for (const row of rows) { + const control = held[`blind-continue::${row.rowId}`] + if (control === undefined) continue + const key = `gated-continue::${row.rowId}` + if (held[key] !== undefined) { + pool += control.promptTokens + control.completionTokens - (held[key]!.promptTokens + held[key]!.completionTokens) + continue + } + if (spent() > COST_CEILING_USD) { + log(`cost ceiling ${COST_CEILING_USD} reached in gated-continue`) + break + } + const entitlement = control.promptTokens + control.completionTokens + const perStep = entitlement / Math.max(1, control.steps.length) + const extra = perStep > 0 ? Math.max(0, Math.floor(pool / perStep)) : 0 + const result = await runRow({ + row, + task: tasks.get(row.taskName)!, + arm: 'gated-continue', + allotment: STEP_ALLOTMENT + extra, + stopOnPass: true, + log, + }) + held[key] = result + pool += entitlement - (result.promptTokens + result.completionTokens) + log(`${row.rowId} gated extra=${extra} pool=${Math.round(pool)}`) + persist() + } + + // ── The registered contrast. The estimand reads `passed` as a number, + // so the boolean verdict is written as 1 or 0 here. + const armRows = (arm: string, passedOf: (run: RowRun) => boolean): EvidenceRecord[] => + rows + .map((row) => held[`${arm}::${row.rowId}`]) + .filter((run): run is RowRun => run !== undefined) + .map((run) => ({ + rowId: run.rowId, + taskName: run.taskName, + arm: arm === 'blind-continue' ? 'blind-continue' : 'gated-continue', + passed: passedOf(run) ? 1 : 0, + })) + const finalPassed = (run: RowRun): boolean => run.steps.at(-1)?.gradePassed === true + const treatment = armRows('gated-continue', (run) => run.bestPassed) + + const contrasts: Record = {} + for (const [label, controlPassed] of [ + ['C-vs-B-best', (run: RowRun) => run.bestPassed], + ['C-vs-B-final', finalPassed], + ] as const) { + const evidence = [...armRows('blind-continue', controlPassed), ...treatment] + contrasts[label] = { + estimand: registered.estimate('pairedContrast', evidence), + interval: registered.interval('pairedContrast95', { kind: 'rows', rows: evidence, value: 'passed' }), + } + } + + const realized = (arm: string): number => + rows + .map((row) => held[`${arm}::${row.rowId}`]) + .filter((run): run is RowRun => run !== undefined) + .reduce((total, run) => total + run.promptTokens + run.completionTokens, 0) + const budgets = registered.matchedBudgets([ + { armId: 'blind-continue', realizedTokens: realized('blind-continue') }, + { armId: 'gated-continue', realizedTokens: realized('gated-continue') }, + ]) + + const servedModels = [...new Set(Object.values(held).flatMap((run) => run.steps.map((s) => s.servedModel)))] + const report = { + generatedAt: new Date().toISOString(), + sealDigest: sealed.digest, + previousSealDigest, + rows: rows.length, + clusters: clusters.length, + servedModels, + contrasts, + budgets, + costUsd: spent(), + settlingN: settling, + } + writeFileSync(join(WORK, 'confirm-report.json'), `${JSON.stringify(report, null, 2)}\n`) + log(`confirm done rows=${rows.length} costUsd=${spent().toFixed(4)} matched=${budgets.matched}`) + return + } + throw new Error(`unknown mode '${mode}'`) }