From c63537073995405633035b9cbca0673d4adf85cf Mon Sep 17 00:00:00 2001 From: Drew Stone Date: Fri, 14 Aug 2026 07:05:12 -0600 Subject: [PATCH 1/4] feat(trace-repair): size the gated-stop draw on its power curve and abort on a substituted model Choose the confirmatory draw on the power curve instead of taking every admitted row. `settlingDraw` returns the smallest draw that clears the registered target at the settling effect of 0.10. The admitted set is the ceiling, not the draw. Near the floor the search estimate and the registered gate estimate straddle the target, because the standard error at 3000 trials is about 0.007 and adjacent draws differ by less. A draw must clear the target at 15000 trials and at the registered parameters before the search accepts it. This selects 151 rows over 22 clusters at power 0.8009 and 0.8013, against 216 admitted rows at 0.9189. Write the confirmatory executor. `confirm` runs the blind arm over the draw, then runs the gated arm one row at a time, returning budget freed by an early clear to a pool that pays for extra steps on rows still failing. Entitlement per row is what the blind arm spent on the same row, so the arms match on realized tokens rather than on steps. The control arm is graded at its final state and at its best intermediate state, and both contrasts are reported. Evaluate the registered `servedModel` gate before grading a row. The seat answers glm-5.2 with glm-5.3, so the gate fails and its registered action aborts the run at zero spend. The refusal records the pinned id, the served id, the rows graded and the dollars spent. Record the measured seat behaviour that bounds the study: the glm-5.x ids are collapsed onto glm-5.3, and temperature 0 does not repeat a reply on either served model. Correct the continuation policy doc, which claimed it did. --- .../gated-stop-ab/confirm-refusal.json | 17 + .../trace-repair/gated-stop-ab/design.json | 275 ++++++++----- docs/trace-repair-continuation.md | 2 +- docs/trace-repair-gated-stop.md | 74 ++++ scripts/tb-gated-stop-ab.ts | 373 +++++++++++++++++- 5 files changed, 633 insertions(+), 108 deletions(-) create mode 100644 benchmarks/trace-repair/gated-stop-ab/confirm-refusal.json create mode 100644 docs/trace-repair-gated-stop.md diff --git a/benchmarks/trace-repair/gated-stop-ab/confirm-refusal.json b/benchmarks/trace-repair/gated-stop-ab/confirm-refusal.json new file mode 100644 index 00000000..2971c0d9 --- /dev/null +++ b/benchmarks/trace-repair/gated-stop-ab/confirm-refusal.json @@ -0,0 +1,17 @@ +{ + "generatedAt": "2026-08-14T13:02:03.856Z", + "sealDigest": "ad82d3424f9949a53f12f2b0a709d0acb6b61f6e6472ba15727b1667c2ad1b4d", + "gate": { + "id": "servedModel", + "passed": false, + "evidence": { + "pinned": "glm-5.2", + "served": "glm-5.3", + "matched": false + } + }, + "onFail": "abort", + "spentUsd": 0, + "rowsGraded": 0, + "note": "The seat answered the pinned model id with a different model. No row was graded and no arm was run. The contrast is not refused on its statistics; it was never started." +} diff --git a/benchmarks/trace-repair/gated-stop-ab/design.json b/benchmarks/trace-repair/gated-stop-ab/design.json index f1f656b9..f76d94f4 100644 --- a/benchmarks/trace-repair/gated-stop-ab/design.json +++ b/benchmarks/trace-repair/gated-stop-ab/design.json @@ -1,7 +1,9 @@ { - "generatedAt": "2026-08-14T12:02:56.974Z", - "sealDigest": "dadf75160af3522b120fc8fb6ba2437c4ab9f460243711db651ebe8939b3899f", - "sealedAt": "2026-08-14T12:02:43.896Z", + "generatedAt": "2026-08-14T13:02:01.887Z", + "previousSealDigest": "dadf75160af3522b120fc8fb6ba2437c4ab9f460243711db651ebe8939b3899f", + "sealDigest": "ad82d3424f9949a53f12f2b0a709d0acb6b61f6e6472ba15727b1667c2ad1b4d", + "sealedAt": "2026-08-14T13:01:50.273Z", + "admittedRows": 216, "clusters": [ { "taskName": "bn-fit-modify", @@ -23,27 +25,7 @@ "break-filter-js-from-html::break-filter-js-from-html__CVHv5bQ::ord0", "break-filter-js-from-html::break-filter-js-from-html__DgW5Uf8::85311391-52de-4291-9216-9d3354e9524d", "break-filter-js-from-html::break-filter-js-from-html__FS9o6H2::ord0", - "break-filter-js-from-html::break-filter-js-from-html__FbAThtu::5e2dd4a5-c0d5-48ac-ae16-a8ed1a89d5d5", - "break-filter-js-from-html::break-filter-js-from-html__HpxC3Nh::ord0", - "break-filter-js-from-html::break-filter-js-from-html__MsAGmXG::9cfa3c39-dd67-4002-8d57-a015ba1e12be", - "break-filter-js-from-html::break-filter-js-from-html__Ren9gY5::ord0", - "break-filter-js-from-html::break-filter-js-from-html__SBPD7KA::ord0", - "break-filter-js-from-html::break-filter-js-from-html__SmkYDSx::869c858f-45d0-481a-91d6-3cb8c235ef5f", - "break-filter-js-from-html::break-filter-js-from-html__UPtYZ3K::ord0", - "break-filter-js-from-html::break-filter-js-from-html__YrKE4JG::a068c57e-6aeb-436f-b4bc-f76a920ab9af", - "break-filter-js-from-html::break-filter-js-from-html__YytgLub::ord0", - "break-filter-js-from-html::break-filter-js-from-html__Z8DKsYa::b0ebf23d-ba67-4458-be9b-10ae8bfea2a1", - "break-filter-js-from-html::break-filter-js-from-html__ZNEGRBE::ord0", - "break-filter-js-from-html::break-filter-js-from-html__auronz2::ord0", - "break-filter-js-from-html::break-filter-js-from-html__cRcDUvb::ord0", - "break-filter-js-from-html::break-filter-js-from-html__eCxehgs::4098e56d-1f5a-468e-95cc-f6232d75e350", - "break-filter-js-from-html::break-filter-js-from-html__jAgdJGf::9a9ff662-1d3c-4794-8a45-f5acfd1c3396", - "break-filter-js-from-html::break-filter-js-from-html__kyKMFmj::ord0", - "break-filter-js-from-html::break-filter-js-from-html__mXWdn3b::ord0", - "break-filter-js-from-html::break-filter-js-from-html__orBRHGT::ord0", - "break-filter-js-from-html::break-filter-js-from-html__y8FbRrP::ord0", - "break-filter-js-from-html::break-filter-js-from-html__yNGimWW::ord0", - "break-filter-js-from-html::break-filter-js-from-html__yQdRrCj::ord0" + "break-filter-js-from-html::break-filter-js-from-html__FbAThtu::5e2dd4a5-c0d5-48ac-ae16-a8ed1a89d5d5" ] }, { @@ -57,11 +39,7 @@ "chess-best-move::chess-best-move__TpjiHFA::895aac58-188e-47c1-965d-754f62f79209", "chess-best-move::chess-best-move__Ue6gFS9::e6461331-38da-480b-9ac0-bc0102f35fe2", "chess-best-move::chess-best-move__XXLV2V6::f57cb929-fbea-4258-8eca-8beb49faf445", - "chess-best-move::chess-best-move__fow4qGA::ord0", - "chess-best-move::chess-best-move__kBK7T4E::ord0", - "chess-best-move::chess-best-move__kzCwv7q::ord0", - "chess-best-move::chess-best-move__nTSQzKk::d381e3f2-5849-469f-a706-b7395f2dc549", - "chess-best-move::chess-best-move__xxMUxUm::014f4b8b-cffd-49c3-8177-6da5ce3c5053" + "chess-best-move::chess-best-move__fow4qGA::ord0" ] }, { @@ -116,12 +94,7 @@ "filter-js-from-html::filter-js-from-html__S8a6ZYA::ord0", "filter-js-from-html::filter-js-from-html__aJouy8R::ord0", "filter-js-from-html::filter-js-from-html__anQHv5W::ord0", - "filter-js-from-html::filter-js-from-html__bck9ykA::a035a05a-2a53-4097-9ef3-e952f27b0993", - "filter-js-from-html::filter-js-from-html__drLu5nC::69319b63-2961-4d55-9394-f9030e2b8913", - "filter-js-from-html::filter-js-from-html__hbv6wH4::a52c0636-3f72-4363-9552-bb42582b7256", - "filter-js-from-html::filter-js-from-html__mHra9p6::93c17aad-f813-421f-9bbe-627d9d876b48", - "filter-js-from-html::filter-js-from-html__w9VS6fN::0f03f8db-3096-4f9c-b1b2-ea1829cb4a89", - "filter-js-from-html::filter-js-from-html__xyoCYLC::7170d09e-137f-417b-8f81-4b4b0e61ea7e" + "filter-js-from-html::filter-js-from-html__bck9ykA::a035a05a-2a53-4097-9ef3-e952f27b0993" ] }, { @@ -147,12 +120,7 @@ "git-leak-recovery::git-leak-recovery__HBnLM3i::97e938e0-f694-4fc3-83bd-35c16ba74dc1", "git-leak-recovery::git-leak-recovery__K8ok2pK::66deae27-26f1-406d-b75f-932657ce72b1", "git-leak-recovery::git-leak-recovery__Nn5ttes::ord0", - "git-leak-recovery::git-leak-recovery__P7SuWyx::6439ad28-caaf-4d56-bb72-076c46dd50d1", - "git-leak-recovery::git-leak-recovery__Tg6AxNw::89d37d5d-c829-4fde-b139-0bc945509939", - "git-leak-recovery::git-leak-recovery__dXFx8P6::ord0", - "git-leak-recovery::git-leak-recovery__qSvc6zx::fc641256-e926-495f-a162-ff0fd13f3824", - "git-leak-recovery::git-leak-recovery__t4LRbnc::ord0", - "git-leak-recovery::git-leak-recovery__vFWKCQt::25641162-1145-4793-8064-abae8e279aa2" + "git-leak-recovery::git-leak-recovery__P7SuWyx::6439ad28-caaf-4d56-bb72-076c46dd50d1" ] }, { @@ -176,9 +144,7 @@ "kv-store-grpc::kv-store-grpc__Nyh8Smx::ord0", "kv-store-grpc::kv-store-grpc__PKKLwCR::1bde1855-361f-4fdb-aa54-ae7edf056417", "kv-store-grpc::kv-store-grpc__UXseWZi::022cc20b-dcec-4afa-8693-ceead1cc010d", - "kv-store-grpc::kv-store-grpc__kdqybAG::57e0f2cf-779e-48d8-9fb7-704cdd08d424", - "kv-store-grpc::kv-store-grpc__qnBaVxn::3308bda2-a678-4a27-8a65-e535b4af0caf", - "kv-store-grpc::kv-store-grpc__xwK4TCS::ord0" + "kv-store-grpc::kv-store-grpc__kdqybAG::57e0f2cf-779e-48d8-9fb7-704cdd08d424" ] }, { @@ -191,11 +157,7 @@ "model-extraction-relu-logits::model-extraction-relu-logits__ZZEsB4i::22b1019d-eee3-4694-be20-b923ae552598", "model-extraction-relu-logits::model-extraction-relu-logits__fqrGcZC::f58041eb-283a-4852-b372-02376394c21a", "model-extraction-relu-logits::model-extraction-relu-logits__hr7cisz::bfb8a27d-fd92-4611-b23b-1a91b0f616e4", - "model-extraction-relu-logits::model-extraction-relu-logits__nFWS2qi::ord0", - "model-extraction-relu-logits::model-extraction-relu-logits__ncK3AWP::07fdced1-9b3a-41bf-8599-738384c87174", - "model-extraction-relu-logits::model-extraction-relu-logits__notJ6hQ::5a0b141a-b512-4a6e-9a66-417c3bf02c0a", - "model-extraction-relu-logits::model-extraction-relu-logits__svNbmWu::ord0", - "model-extraction-relu-logits::model-extraction-relu-logits__uWYzbk8::ord0" + "model-extraction-relu-logits::model-extraction-relu-logits__nFWS2qi::ord0" ] }, { @@ -242,13 +204,7 @@ "path-tracing::path-tracing__KYYhxrN::ord0", "path-tracing::path-tracing__Ld2biwh::ord0", "path-tracing::path-tracing__N9ENVmn::ord0", - "path-tracing::path-tracing__P9ZEiuh::4cb5116e-7262-4d5b-9b58-467c65e8e40f", - "path-tracing::path-tracing__Z4hmti2::ord0", - "path-tracing::path-tracing__feCVATm::ord0", - "path-tracing::path-tracing__ppNTLQZ::93f5b4a7-2cfe-4b6d-93e8-942afc0ccfd4", - "path-tracing::path-tracing__qVQgJUh::bc8bac3d-4808-4d65-9471-6daccb42cce6", - "path-tracing::path-tracing__utnFhM5::ord0", - "path-tracing::path-tracing__vtQ5md9::ord0" + "path-tracing::path-tracing__P9ZEiuh::4cb5116e-7262-4d5b-9b58-467c65e8e40f" ] }, { @@ -268,8 +224,7 @@ "qemu-startup::qemu-startup__dxi9J6K::ord0", "qemu-startup::qemu-startup__fHCKqAu::ord0", "qemu-startup::qemu-startup__j52j9ou::ord0", - "qemu-startup::qemu-startup__jAgPZ7R::ord0", - "qemu-startup::qemu-startup__yidXM4m::ord0" + "qemu-startup::qemu-startup__jAgPZ7R::ord0" ] }, { @@ -282,11 +237,7 @@ "regex-chess::regex-chess__NB8Tki9::27fee60b-cc7b-4208-883a-93736ee8263a", "regex-chess::regex-chess__ZoA7qpB::ord0", "regex-chess::regex-chess__ZynBG3r::d239da8a-4f1f-4178-b744-aa2f6697d830", - "regex-chess::regex-chess__cPycyuX::ord0", - "regex-chess::regex-chess__gGEmXrS::ord0", - "regex-chess::regex-chess__gzsTxh7::ord0", - "regex-chess::regex-chess__n9wfYsh::44a65548-1325-469b-b86c-7fa01cbf2b3d", - "regex-chess::regex-chess__tipe2JT::ef65d19c-e746-46d0-9d81-670cf5a35131" + "regex-chess::regex-chess__cPycyuX::ord0" ] }, { @@ -299,18 +250,7 @@ "regex-log::regex-log__KK3RP5S::0025f241-a253-4019-a1f8-cd9d9c731b28", "regex-log::regex-log__UGzQbG8::ord0", "regex-log::regex-log__V7P89ps::59942205-d29e-46fa-993e-96918c5ba9f9", - "regex-log::regex-log__WQxJoLA::ord0", - "regex-log::regex-log__Ymz5e27::ord0", - "regex-log::regex-log__ZLpgtxv::ord0", - "regex-log::regex-log__bQtFPoF::2c22649d-e689-4373-a04e-eb4febdc6b0b", - "regex-log::regex-log__eo3uVSo::f015bde3-ac57-45ba-85f3-6fa03fae3a76", - "regex-log::regex-log__erETYKP::ord0", - "regex-log::regex-log__fMAABjG::41189d53-6fac-4ed2-97be-cd7a4d45963b", - "regex-log::regex-log__mZCzu9B::ord0", - "regex-log::regex-log__nbpDq4C::ord0", - "regex-log::regex-log__qZszGwS::56c940d6-ba36-42b0-88a5-8e4362ed5156", - "regex-log::regex-log__xnmvrp7::f040ead1-261a-47ee-ab1e-5fff3202aedf", - "regex-log::regex-log__yoY6Ds3::ord0" + "regex-log::regex-log__WQxJoLA::ord0" ] }, { @@ -323,15 +263,12 @@ "winning-avg-corewars::winning-avg-corewars__Z2SbZS5::ord0", "winning-avg-corewars::winning-avg-corewars__ZEpUJJ8::dd3d130e-5280-4fef-953f-621fc41e3e57", "winning-avg-corewars::winning-avg-corewars__bXfu7mQ::b62bb781-2411-4b62-b9df-89b5d358a76f", - "winning-avg-corewars::winning-avg-corewars__gzJDGeV::ord0", - "winning-avg-corewars::winning-avg-corewars__m9YyonB::7f021107-703a-48e1-bd1f-96ab69ff69bb", - "winning-avg-corewars::winning-avg-corewars__mqqT9Wz::29e431fc-70da-4ad0-a561-596b067c0a35", - "winning-avg-corewars::winning-avg-corewars__weYHwi9::5bb2ba2e-5885-4270-8499-1c8735ea21b9" + "winning-avg-corewars::winning-avg-corewars__gzJDGeV::ord0" ] } ], "clusterCount": 22, - "rowCount": 216, + "rowCount": 151, "funnel": { "population": "terminalbench-trajectories, mini-swe-agent rows with reward 0, on tasks whose oracle certification is checked in", "input": 2601, @@ -414,7 +351,7 @@ ], "power": { "clusterCount": 22, - "totalRows": 216, + "totalRows": 151, "trials": 3000, "resamples": 3000, "seed": 20260814, @@ -423,26 +360,26 @@ "curve": [ { "effect": 0.05, - "power": 0.494, - "medianCiWidth": 0.09744285160790392 + "power": 0.3453333333333333, + "medianCiWidth": 0.11652017128690556 }, { "effect": 0.1, - "power": 0.9236666666666666, - "medianCiWidth": 0.11107147756893951 + "power": 0.8013333333333333, + "medianCiWidth": 0.132921528254404 }, { "effect": 0.15, - "power": 0.9943333333333333, - "medianCiWidth": 0.12135501355013549 + "power": 0.972, + "medianCiWidth": 0.1461789844142785 }, { "effect": 0.2, - "power": 1, - "medianCiWidth": 0.1303355299995693 + "power": 0.9956666666666667, + "medianCiWidth": 0.15600506875674658 } ], - "maxPower": 1, + "maxPower": 0.9956666666666667, "signFlipFloor": { "twoSidedP": 4.76837158203125e-7, "oneSidedP": 2.384185791015625e-7, @@ -458,23 +395,23 @@ "passed": true, "evidence": { "target": 0.8, - "maxPower": 1, + "maxPower": 0.9956666666666667, "curve": [ { "effect": 0.05, - "power": 0.494 + "power": 0.3453333333333333 }, { "effect": 0.1, - "power": 0.9236666666666666 + "power": 0.8013333333333333 }, { "effect": 0.15, - "power": 0.9943333333333333 + "power": 0.972 }, { "effect": 0.2, - "power": 1 + "power": 0.9956666666666667 } ] } @@ -484,5 +421,151 @@ "action": null, "failedGates": [] }, - "settlingN": null + "settlingN": { + "effect": 0.1, + "target": 0.8, + "trials": 15000, + "rows": 151, + "power": 0.8008666666666666, + "powerAtRegisteredSim": 0.8013333333333333, + "maxRows": 216, + "maxPower": 0.9188666666666667, + "scanned": [ + { + "rows": 22, + "power": 0.1708, + "registered": null + }, + { + "rows": 44, + "power": 0.3263333333333333, + "registered": null + }, + { + "rows": 66, + "power": 0.46, + "registered": null + }, + { + "rows": 88, + "power": 0.5688666666666666, + "registered": null + }, + { + "rows": 110, + "power": 0.6672, + "registered": null + }, + { + "rows": 132, + "power": 0.7476, + "registered": null + }, + { + "rows": 133, + "power": 0.7504666666666666, + "registered": null + }, + { + "rows": 134, + "power": 0.7564666666666666, + "registered": null + }, + { + "rows": 135, + "power": 0.7578666666666667, + "registered": null + }, + { + "rows": 136, + "power": 0.7530666666666667, + "registered": null + }, + { + "rows": 137, + "power": 0.7691333333333333, + "registered": null + }, + { + "rows": 138, + "power": 0.7695333333333333, + "registered": null + }, + { + "rows": 139, + "power": 0.7699333333333334, + "registered": null + }, + { + "rows": 140, + "power": 0.7684666666666666, + "registered": null + }, + { + "rows": 141, + "power": 0.7780666666666667, + "registered": null + }, + { + "rows": 142, + "power": 0.7804, + "registered": null + }, + { + "rows": 143, + "power": 0.7860666666666667, + "registered": null + }, + { + "rows": 144, + "power": 0.781, + "registered": null + }, + { + "rows": 145, + "power": 0.7913333333333333, + "registered": null + }, + { + "rows": 146, + "power": 0.7876666666666666, + "registered": null + }, + { + "rows": 147, + "power": 0.7983333333333333, + "registered": null + }, + { + "rows": 148, + "power": 0.7947333333333333, + "registered": null + }, + { + "rows": 149, + "power": 0.8055333333333333, + "registered": 0.7896666666666666 + }, + { + "rows": 150, + "power": 0.7961333333333334, + "registered": null + }, + { + "rows": 151, + "power": 0.8008666666666666, + "registered": 0.8013333333333333 + }, + { + "rows": 154, + "power": 0.8090666666666667, + "registered": 0.814 + }, + { + "rows": 216, + "power": 0.9188666666666667, + "registered": null + } + ] + } } diff --git a/docs/trace-repair-continuation.md b/docs/trace-repair-continuation.md index b79b61da..e32febd7 100644 --- a/docs/trace-repair-continuation.md +++ b/docs/trace-repair-continuation.md @@ -15,7 +15,7 @@ The pre-pass that decides which rows the arms run on is in [trace-repair-admissi | --- | --- | --- | | scaffold | mini-swe-agent | The corpus recorded it, so a continuation stays in the same distribution as the prefix. | | step budget | 20 model calls | Bounds a rollout without a wall-clock limit, which would end rollouts at different points. | -| temperature | 0 | With a fixed seed, the same prefix draws the same continuation. | +| temperature | 0 | Removes the sampler as a source of variation the policy controls. It does not make a continuation repeat: measured against the z.ai seat, 19 of 20 replies to one identical prompt were distinct on `glm-5.3` and 8 of 20 on `glm-4.7`. A paired design must carry that variation as a threat to validity — see [trace-repair-gated-stop.md](./trace-repair-gated-stop.md). | | command timeout | 30 s | The recorded runs used the scaffold's own 30-second limit. A longer limit lets the continuation finish commands the recorded agent could not. | | network | `none` | A container with a network can install what the recorded run could not. | | format-error cap | 3 consecutive turns | Ends a rollout that has stopped producing actions. | diff --git a/docs/trace-repair-gated-stop.md b/docs/trace-repair-gated-stop.md new file mode 100644 index 00000000..b93b1ab9 --- /dev/null +++ b/docs/trace-repair-gated-stop.md @@ -0,0 +1,74 @@ +# Gated stop against blind continuation + +The study asks one question. +At one matched total token budget, does an agent that may stop only after an executable held-out check passes finish more rows than the same agent spending the identical budget on unconditional continuation? + +The mechanism under test is budget allocation, not detection. +A row that clears the check early returns its unspent budget to a pool. +The pool pays for extra steps on rows that have not cleared. +Detection quality is not the claim: on this corpus the recorded done-signal fires on 51.69 % of failed runs and 31.14 % of successes, which is 62.5 % precision as a success predictor. + +Design and runner: [`benchmarks/trace-repair/gated-stop-ab/design.json`](../benchmarks/trace-repair/gated-stop-ab/design.json) and [`scripts/tb-gated-stop-ab.ts`](../scripts/tb-gated-stop-ab.ts). +The registration primitives are described in [experiment.md](./experiment.md). + +## The two arms + +| arm | role | stop rule | graded at | +| --- | --- | --- | --- | +| `blind-continue` | control | spend the allotment unconditionally | best intermediate state | +| `gated-continue` | treatment | stop when the held-out check passes | best intermediate state | + +Both arms replay the same recorded prefix, run the same scaffold, and are graded by the same injected suite after every step. +The check that gates the treatment arm is the check that scores both arms. + +That symmetry forces a control the report must carry. +The gate is the outcome, so the treatment arm cannot lose a success it already reached, while the control arm can regress out of one. +The control arm is therefore graded twice, at its final state and at its best intermediate state, and both contrasts are reported. +When the two disagree, the harsher contrast is the headline. + +## Choosing the draw + +The admitted set is the ceiling, not the draw. +`settlingDraw` returns the smallest draw that clears the registered power target at the settling effect of 0.10. +Spending more rows than the design needs is as undisciplined as spending fewer. + +The search runs at 15 000 trials and the registered gate runs at 3 000. +Near the floor the two estimates straddle the target, because the standard error at 3 000 trials is about 0.007 and adjacent draws differ by less. +A draw must clear the target under both before the search accepts it. + +## The identity gate runs before the spend + +`confirm` evaluates the registered `servedModel` gate before it grades a row. +The gate compares the pinned model id against the id the seat reports. +Its registered action on failure is `abort`. + +A contrast measured on a substituted model belongs to an experiment nobody registered. +The gate therefore ends the run at zero spend and writes `confirm-refusal.json`, which records the pinned id, the served id, the rows graded, and the dollars spent. +A refusal is a verdict object, not prose beside one. + +## Measured seat behaviour + +These facts were measured against the z.ai coding seat and they bound what the study can claim. + +| fact | measurement | n | +| --- | --- | --- | +| `glm-5.2`, `glm-5.1` and `glm-5` are all answered by `glm-5.3` | served id differs from requested id | 3 ids | +| `glm-4.7` and `glm-4.6` are answered by themselves | served id equals requested id | 2 ids | +| `glm-5.3` is not deterministic at temperature 0 | 19 of 20 replies distinct | 20 | +| `glm-4.7` is not deterministic at temperature 0 | 8 of 20 replies distinct | 20 | +| the scaffold's first turn draws more than one action block | 6 of 20 replies, both models | 20 per model | +| the same rate inside a run, behind a replayed prefix | 4 of 24 steps | 24 | + +Temperature 0 does not give a repeated continuation on this provider. +A paired design that assumes it must treat run-to-run variation as a threat to validity and report it. + +## Label quality on `qemu-startup` + +One row, `qemu-startup__PnXK6EH::ord0`, is recorded at reward 0 and passes its own held-out suite on the replayed end state. +The end-state screen measures exactly that condition, so the registered funnel excludes the row at the `recorded-end-state-fails-its-own-suite` stage. +The row is absent from the draw. + +The disagreement rate is 1 of 18 screened rows on `qemu-startup`, against 0 of 285 on every other task. +The concentration is the finding, not the single row. +`qemu-startup` rows still enter the study, because the screen tests each row against the condition that would disqualify it and the remaining rows pass that test. +A task whose labels disagree with its own oracle at 5.6 % cannot carry a study on its own, and this design does not ask it to: it contributes 8 of 151 rows inside a task-clustered bootstrap that resamples whole tasks. diff --git a/scripts/tb-gated-stop-ab.ts b/scripts/tb-gated-stop-ab.ts index 4666bdbd..daa9b576 100644 --- a/scripts/tb-gated-stop-ab.ts +++ b/scripts/tb-gated-stop-ab.ts @@ -32,15 +32,19 @@ * The power floor is a registered gate, so a structure that cannot see the * effect refuses the spend instead of producing an interval nobody can read. * - * node --import tsx scripts/tb-gated-stop-ab.ts design + * The confirmatory draw is chosen on the power curve, not on appetite: the + * admitted set is the ceiling, and `settlingDraw` returns the smallest draw + * that clears the registered target at the settling effect. + * + * `confirm` evaluates the registered identity gate before it grades a row. A + * seat that answers the pinned model id with another model aborts the run at + * zero spend, because a contrast measured on a substituted model belongs to an + * experiment nobody registered. + * + * node --import tsx scripts/tb-gated-stop-ab.ts design [--take N] * node --import tsx scripts/tb-gated-stop-ab.ts screen * node --import tsx scripts/tb-gated-stop-ab.ts pilot --rows 6 --steps 8 - * - * The confirmatory executor that runs both arms is not written. The design's - * power gate refused the structure it was registered against, so nothing could - * have executed it; the recovered corpus clears that gate, and building the - * executor is the next piece of work rather than something this file already - * does. `stopOnPass` exists for it and no current mode passes `true`. + * node --import tsx scripts/tb-gated-stop-ab.ts confirm */ import { execFile } from 'node:child_process' @@ -230,6 +234,148 @@ const REGISTERED_EFFECT_GRID = [0.05, 0.1, 0.15, 0.2] const POWER_TARGET = 0.8 const POWER_SIM = { trials: 3000, resamples: 3000, seed: 20260814 } +/** + * The effect the confirmatory draw is sized against. + * + * 0.05 is below what this mechanism can generate and 0.15 would buy a draw too + * small to read a plausible effect, so the grid's second point is the one the + * row count is chosen on. + */ +const SETTLING_EFFECT = 0.1 +/** + * Trials for the settling search only. + * + * The registered gate is judged at `POWER_SIM`. Choosing the row count on that + * many trials would pick a draw whose margin over the target is Monte Carlo + * luck: at 3000 trials the standard error near 0.8 is about 0.007, which is + * wider than the gap between adjacent draws. The search runs hotter, and the + * chosen draw is then re-checked at the registered parameters. + */ +const SETTLING_TRIALS = 15_000 + +/** + * Rows each cluster contributes when the registered round-robin takes `take`. + * + * The selection cycles tasks in lexical order and takes one row per pass, so + * the structure a smaller draw carries follows from the registered rule rather + * than from a second rule invented for the search. + */ +function roundRobinSizes( + clusters: readonly { taskName: string; rowIds: readonly string[] }[], + take: number, +): number[] { + const ordered = [...clusters].sort((a, b) => a.taskName.localeCompare(b.taskName)) + const counts = new Map() + let taken = 0 + for (let pass = 0; taken < take; pass += 1) { + let progressed = false + for (const cluster of ordered) { + if (taken >= take) break + if (pass >= cluster.rowIds.length) continue + counts.set(cluster.taskName, (counts.get(cluster.taskName) ?? 0) + 1) + taken += 1 + progressed = true + } + if (!progressed) break + } + return [...counts.values()] +} + +function powerAt(sizes: readonly number[], effect: number, trials: number): number { + return clusteredPower({ + clusterSizes: [...sizes], + effects: [effect], + seed: POWER_SIM.seed, + trials, + resamples: POWER_SIM.resamples, + targetPower: POWER_TARGET, + baseWinRate: 0.05, + baseLossRate: 0.05, + }).curve[0]!.power +} + +export interface SettlingDraw { + effect: number + target: number + trials: number + /** Smallest draw whose power clears the target at `effect`. */ + rows: number + power: number + /** The same draw re-checked at the registered gate's simulation parameters. */ + powerAtRegisteredSim: number + /** The largest draw searched, and its power — the ceiling the corpus offers. */ + maxRows: number + maxPower: number + /** Every draw the search evaluated, so the curve is auditable. */ + scanned: { rows: number; power: number; registered: number | null }[] +} + +/** + * The smallest confirmatory draw that clears the power target. + * + * Taking every admitted row spends rows the design does not need. The search + * walks the draw upward and returns the first that clears, so the run buys the + * power the design registered and nothing beyond it. + */ +function settlingDraw( + clusters: readonly { taskName: string; rowIds: readonly string[] }[], + maxTake: number, +): SettlingDraw { + const clusterCount = clusters.length + const scanned: { rows: number; power: number; registered: number | null }[] = [] + const evaluate = (rows: number): number => { + const found = scanned.find((point) => point.rows === rows) + if (found !== undefined) return found.power + const power = powerAt(roundRobinSizes(clusters, rows), SETTLING_EFFECT, SETTLING_TRIALS) + scanned.push({ rows, power, registered: null }) + return power + } + /** + * The registered gate re-check. + * + * The search runs hotter than the gate, so near the floor the two estimates + * straddle the target. A draw chosen only on the hotter estimate would carry + * a gate that reads below the floor on the same draw, so a draw must clear + * the target under both before it is chosen. + */ + const registeredCheck = (rows: number): number => { + const point = scanned.find((entry) => entry.rows === rows)! + if (point.registered === null) { + point.registered = powerAt(roundRobinSizes(clusters, rows), SETTLING_EFFECT, POWER_SIM.trials) + } + return point.registered + } + const clears = (rows: number): boolean => + evaluate(rows) >= POWER_TARGET && registeredCheck(rows) >= POWER_TARGET + // Bracket on a coarse grid, then refine by one row at a time. + let bracket = maxTake + for (let rows = clusterCount; rows <= maxTake; rows += clusterCount) { + if (clears(rows)) { + bracket = rows + break + } + } + let chosen = bracket + for (let rows = Math.max(clusterCount, bracket - clusterCount + 1); rows <= bracket; rows += 1) { + if (clears(rows)) { + chosen = rows + break + } + } + scanned.sort((a, b) => a.rows - b.rows) + return { + effect: SETTLING_EFFECT, + target: POWER_TARGET, + trials: SETTLING_TRIALS, + rows: chosen, + power: evaluate(chosen), + powerAtRegisteredSim: registeredCheck(chosen), + maxRows: maxTake, + maxPower: evaluate(maxTake), + scanned, + } +} + function buildSpec(certifiedTasks: string[], sealedRowIds: string[], take: number, pilotExposed: string[]) { return defineExperiment({ id: 'gated-stop-vs-blind-continue-tb-repair-v1', @@ -858,9 +1004,27 @@ async function main(): Promise { await sealExperiment(buildSpec(design.certifiedTasks, [], design.take, exposed)), ) const admitted = draft.admit(design.records) - const drawn = draft.select('confirmatory', admitted.survivors, { idField: 'rowId' }) + const admittedDraw = draft.select('confirmatory', admitted.survivors, { idField: 'rowId' }) - const sealed = await sealExperiment(buildSpec(design.certifiedTasks, drawn, design.take, exposed)) + // The admitted set is the ceiling, not the draw. The row count is chosen on + // the power curve: the smallest draw that clears the registered target at + // the settling effect. + const byTask = new Map() + for (const id of admittedDraw) { + const task = id.split('::')[0]! + byTask.set(task, [...(byTask.get(task) ?? []), id]) + } + const admittedClusters = [...byTask.entries()].map(([taskName, rowIds]) => ({ taskName, rowIds })) + const settling = settlingDraw(admittedClusters, admittedDraw.length) + const takeArg = process.argv.indexOf('--take') + const take = takeArg > 0 ? Number(process.argv[takeArg + 1]) : settling.rows + + const sized = await openSealedExperiment( + await sealExperiment(buildSpec(design.certifiedTasks, [], take, exposed)), + ) + const drawn = sized.select('confirmatory', admitted.survivors, { idField: 'rowId' }) + + const sealed = await sealExperiment(buildSpec(design.certifiedTasks, drawn, take, exposed)) const registered = await openSealedExperiment(sealed) const redrawn = registered.select('confirmatory', registered.admit(design.records).survivors, { idField: 'rowId', @@ -891,10 +1055,25 @@ async function main(): Promise { }) const halt = registered.halt([powerGate]) + // The digest this draw replaces, read from the checked-in design so the + // report carries the chain rather than a single current hash. + const previousSealDigest = ((): string | null => { + try { + const held = JSON.parse( + readFileSync(join(import.meta.dirname, '..', 'benchmarks', 'trace-repair', 'gated-stop-ab', 'design.json'), 'utf8'), + ) as { sealDigest?: string } + return held.sealDigest ?? null + } catch { + return null + } + })() + const designReport = { generatedAt: new Date().toISOString(), + previousSealDigest, sealDigest: sealed.digest, sealedAt: sealed.sealedAt, + admittedRows: admittedDraw.length, clusters, clusterCount: clusters.length, rowCount: chosen.length, @@ -903,12 +1082,15 @@ async function main(): Promise { power, powerGate, halt, - settlingN: null as unknown, + settlingN: settling, } writeFileSync(join(WORK, 'design.json'), `${JSON.stringify(designReport, null, 2)}\n`) process.stdout.write( - `seal=${sealed.digest}\nclusters=${clusters.length} rows=${chosen.length}\n` + + `previousSeal=${previousSealDigest ?? 'none'}\nseal=${sealed.digest}\n` + + `clusters=${clusters.length} rows=${chosen.length} (admitted ${admittedDraw.length})\n` + `funnel: ${admitted.funnel.input} -> ${admitted.funnel.surviving} admitted -> ${chosen.length} drawn\n` + + `settling: rows=${settling.rows} power@${settling.effect}=${settling.power.toFixed(4)} registeredSim=${settling.powerAtRegisteredSim.toFixed(4)} ` + + `(ceiling rows=${settling.maxRows} power=${settling.maxPower.toFixed(4)})\n` + `power: ${power.curve.map((p) => `${p.effect}=>${p.power.toFixed(2)}`).join(' ')}\n` + `gate powerFloor passed=${powerGate.passed}; halt fired=${halt.fired} action=${halt.action ?? 'none'}\n`, ) @@ -1025,6 +1207,175 @@ async function main(): Promise { return } + if (mode === 'confirm') { + const log = logger(join(WORK, 'confirm.log')) + + // The identity gate runs before a row is graded and before a token is + // spent. A seat that answers the pinned id with another model produces a + // contrast for a model nobody registered, so the registered action here is + // to abort rather than to record the substitution and continue. + const probe = await callModel([{ role: 'user', content: 'reply with the single word ok' }]) + const identity = registered.gate('servedModel', { + kind: 'identity', + pinned: MODEL, + served: probe.servedModel, + }) + log(`servedModel gate: pinned=${MODEL} served=${probe.servedModel} passed=${identity.passed}`) + if (!identity.passed) { + const refusal = { + generatedAt: new Date().toISOString(), + sealDigest: sealed.digest, + gate: identity, + onFail: 'abort', + spentUsd: 0, + rowsGraded: 0, + note: + 'The seat answered the pinned model id with a different model. No row was graded and no ' + + 'arm was run. The contrast is not refused on its statistics; it was never started.', + } + writeFileSync(join(WORK, 'confirm-refusal.json'), `${JSON.stringify(refusal, null, 2)}\n`) + process.stdout.write( + `ABORT servedModel: pinned=${MODEL} served=${probe.servedModel}; no spend, no rows graded\n`, + ) + process.exitCode = 3 + return + } + + const rows = chosen + const tasks = new Map() + for (const name of new Set(rows.map((row) => row.taskName))) tasks.set(name, await loadTask(name)) + + // Runs persist as they finish. A transport fault hours in must not force a + // re-spend of everything before it. + const statePath = join(WORK, 'confirm-runs.json') + const held: Record = (() => { + try { + return JSON.parse(readFileSync(statePath, 'utf8')) as Record + } catch { + return {} + } + })() + const persist = (): void => writeFileSync(statePath, `${JSON.stringify(held, null, 2)}\n`) + const spent = (): number => Object.values(held).reduce((total, run) => total + run.costUsd, 0) + + // ── Arm B: blind continuation, the whole allotment, whatever the check says. + let next = 0 + await Promise.all( + Array.from({ length: Math.min(SEAT_CONCURRENCY, rows.length) }, async () => { + for (;;) { + const index = next + next += 1 + if (index >= rows.length) return + const row = rows[index]! + const key = `blind-continue::${row.rowId}` + if (held[key] !== undefined) continue + if (spent() > COST_CEILING_USD) { + log(`cost ceiling ${COST_CEILING_USD} reached in blind-continue`) + return + } + held[key] = await runRow({ + row, + task: tasks.get(row.taskName)!, + arm: 'blind-continue', + allotment: STEP_ALLOTMENT, + stopOnPass: false, + log, + }) + persist() + } + }), + ) + + // ── Arm C: gated continuation, one row at a time in the registered order. + // + // A row that clears the check early returns the rest of its entitlement to + // a pool, and the pool pays for extra steps on rows still failing. The + // entitlement is what the blind arm actually spent on the same row, so the + // arms are matched on realized tokens rather than on steps. + let pool = 0 + for (const row of rows) { + const control = held[`blind-continue::${row.rowId}`] + if (control === undefined) continue + const key = `gated-continue::${row.rowId}` + if (held[key] !== undefined) { + pool += control.promptTokens + control.completionTokens - (held[key]!.promptTokens + held[key]!.completionTokens) + continue + } + if (spent() > COST_CEILING_USD) { + log(`cost ceiling ${COST_CEILING_USD} reached in gated-continue`) + break + } + const entitlement = control.promptTokens + control.completionTokens + const perStep = entitlement / Math.max(1, control.steps.length) + const extra = perStep > 0 ? Math.max(0, Math.floor(pool / perStep)) : 0 + const result = await runRow({ + row, + task: tasks.get(row.taskName)!, + arm: 'gated-continue', + allotment: STEP_ALLOTMENT + extra, + stopOnPass: true, + log, + }) + held[key] = result + pool += entitlement - (result.promptTokens + result.completionTokens) + log(`${row.rowId} gated extra=${extra} pool=${Math.round(pool)}`) + persist() + } + + // ── The registered contrast. + const armRows = (arm: string, passedOf: (run: RowRun) => boolean): EvidenceRecord[] => + rows + .map((row) => held[`${arm}::${row.rowId}`]) + .filter((run): run is RowRun => run !== undefined) + .map((run) => ({ + rowId: run.rowId, + taskName: run.taskName, + arm: arm === 'blind-continue' ? 'blind-continue' : 'gated-continue', + passed: passedOf(run), + })) + const finalPassed = (run: RowRun): boolean => run.steps.at(-1)?.gradePassed === true + const treatment = armRows('gated-continue', (run) => run.bestPassed) + + const contrasts: Record = {} + for (const [label, controlPassed] of [ + ['C-vs-B-best', (run: RowRun) => run.bestPassed], + ['C-vs-B-final', finalPassed], + ] as const) { + const evidence = [...armRows('blind-continue', controlPassed), ...treatment] + contrasts[label] = { + estimand: registered.estimate('pairedContrast', evidence), + interval: registered.interval('pairedContrast95', { kind: 'rows', rows: evidence, value: 'passed' }), + } + } + + const realized = (arm: string): number => + rows + .map((row) => held[`${arm}::${row.rowId}`]) + .filter((run): run is RowRun => run !== undefined) + .reduce((total, run) => total + run.promptTokens + run.completionTokens, 0) + const budgets = registered.matchedBudgets([ + { armId: 'blind-continue', realizedTokens: realized('blind-continue') }, + { armId: 'gated-continue', realizedTokens: realized('gated-continue') }, + ]) + + const servedModels = [...new Set(Object.values(held).flatMap((run) => run.steps.map((s) => s.servedModel)))] + const report = { + generatedAt: new Date().toISOString(), + sealDigest: sealed.digest, + previousSealDigest, + rows: rows.length, + clusters: clusters.length, + servedModels, + contrasts, + budgets, + costUsd: spent(), + settlingN: settling, + } + writeFileSync(join(WORK, 'confirm-report.json'), `${JSON.stringify(report, null, 2)}\n`) + log(`confirm done rows=${rows.length} costUsd=${spent().toFixed(4)} matched=${budgets.matched}`) + return + } + throw new Error(`unknown mode '${mode}'`) } From 4616e01272a3a705a1e62c8cfade100a875acb49 Mon Sep 17 00:00:00 2001 From: Drew Stone Date: Fri, 14 Aug 2026 07:20:59 -0600 Subject: [PATCH 2/4] fix(trace-repair): pin the gated-stop arms to the model the seat serves The seat retired glm-5.2 and aliases it onto glm-5.3. A request that names glm-5.2 is answered by glm-5.3, so the pinned id registered a model the run would never use, and the registered `servedModel` gate correctly aborted the confirmatory run at zero spend. Pin both arms to glm-5.3, which the seat answers as itself. Measured on the seat: glm-5.2 is served by glm-5.3, and glm-5.3 is served by glm-5.3. The gate keeps its absolute form. It compares the id on the reply against the pinned id and aborts on any disagreement. The pin moved to the truth; the comparison did not move. Re-sealing carries the model id in both arm pins, so the digest changes from ad82d342 to e13a4f8d. The draw does not: 151 rows over 22 clusters, power 0.8009 at the settling effect of 0.10, `powerFloor` passed, halt not fired. Raise the operating ceiling to the briefed 40 USD. A ceiling that stops an arm part way leaves rows with a control run and no treatment run, and the registered estimand scores that pair as a zero difference. --- .../trace-repair/gated-stop-ab/design.json | 8 +++---- scripts/tb-gated-stop-ab.ts | 21 ++++++++++++++++--- 2 files changed, 22 insertions(+), 7 deletions(-) diff --git a/benchmarks/trace-repair/gated-stop-ab/design.json b/benchmarks/trace-repair/gated-stop-ab/design.json index f76d94f4..b328b530 100644 --- a/benchmarks/trace-repair/gated-stop-ab/design.json +++ b/benchmarks/trace-repair/gated-stop-ab/design.json @@ -1,8 +1,8 @@ { - "generatedAt": "2026-08-14T13:02:01.887Z", - "previousSealDigest": "dadf75160af3522b120fc8fb6ba2437c4ab9f460243711db651ebe8939b3899f", - "sealDigest": "ad82d3424f9949a53f12f2b0a709d0acb6b61f6e6472ba15727b1667c2ad1b4d", - "sealedAt": "2026-08-14T13:01:50.273Z", + "generatedAt": "2026-08-14T13:20:36.726Z", + "previousSealDigest": "ad82d3424f9949a53f12f2b0a709d0acb6b61f6e6472ba15727b1667c2ad1b4d", + "sealDigest": "e13a4f8d9ff437dbf7d24898d64cba55c90c682c3902ad30bb528ca241939626", + "sealedAt": "2026-08-14T13:20:28.225Z", "admittedRows": 216, "clusters": [ { diff --git a/scripts/tb-gated-stop-ab.ts b/scripts/tb-gated-stop-ab.ts index daa9b576..2e9da81e 100644 --- a/scripts/tb-gated-stop-ab.ts +++ b/scripts/tb-gated-stop-ab.ts @@ -83,8 +83,15 @@ const WORK = '/home/drew/bench-cache/gated-stop-ab' const ROWS_PATH = join(WORK, 'rows-decoded.json') const TASK_ORACLES_PATH = join(import.meta.dirname, '..', 'benchmarks', 'trace-repair', 'task-oracles.json') -/** z.ai coding plan, OpenAI-compatible. The seat this run is entitled to. */ -const MODEL = 'glm-5.2' +/** + * z.ai coding plan, OpenAI-compatible. The seat this run is entitled to. + * + * The id is pinned to what the seat serves, not to what it accepts. The seat + * accepts the retired glm-5.2 id and answers it with glm-5.3, so pinning the + * retired id registers a model the run never uses. The `servedModel` gate reads + * the id the seat reports on the reply and aborts on any disagreement. + */ +const MODEL = 'glm-5.3' const BASE_URL = 'https://api.z.ai/api/coding/paas/v4/chat/completions' const PRICING = { inputUsdPerMillion: 0.6, outputUsdPerMillion: 2.2 } /** At most three in flight on this seat, so a 429 storm cannot be mistaken for a slow model. */ @@ -94,7 +101,15 @@ const MODEL_DEADLINE_MS = 180_000 const STEP_TIMEOUT_SECONDS = 30 const REPLAY_TIMEOUT_MS = 120_000 const GRADE_TIMEOUT_MS = 420_000 -const COST_CEILING_USD = 25 +/** + * The operating ceiling for the confirmatory run. + * + * A ceiling that stops an arm part way leaves rows with a control run and no + * treatment run, and the registered estimand scores that pair as a zero + * difference. That dilutes the contrast toward null, so the ceiling is set + * above the draw's projected spend rather than at it. + */ +const COST_CEILING_USD = 40 /** Per-row model steps the blind arm spends whatever the check says. */ const STEP_ALLOTMENT = 8 From 0dcef77dccb01ff3bacf73303a877fefcd1afbba Mon Sep 17 00:00:00 2001 From: Drew Stone Date: Fri, 14 Aug 2026 07:49:24 -0600 Subject: [PATCH 3/4] docs(trace-repair): record how the gated-stop arms are run and resumed State why the treatment arm is serial. Its pool is sequential state, so processing order decides which rows receive returned budget. Concurrent rows would change the allocation the seal registered. State the resume contract. Every finished row is written before the next row starts, and a re-invocation rebuilds the pool from the rows it already holds. State that the matched-budget rule is a refusal object. Budget freed by the last rows has no later row to pay for, and a trailing shortfall past 5 % returns `contrast-refused-unmatched-budget`. --- docs/trace-repair-gated-stop.md | 27 +++++++++++++++++++++++++++ 1 file changed, 27 insertions(+) diff --git a/docs/trace-repair-gated-stop.md b/docs/trace-repair-gated-stop.md index b93b1ab9..5b7b6a2c 100644 --- a/docs/trace-repair-gated-stop.md +++ b/docs/trace-repair-gated-stop.md @@ -72,3 +72,30 @@ The disagreement rate is 1 of 18 screened rows on `qemu-startup`, against 0 of 2 The concentration is the finding, not the single row. `qemu-startup` rows still enter the study, because the screen tests each row against the condition that would disqualify it and the remaining rows pass that test. A task whose labels disagree with its own oracle at 5.6 % cannot carry a study on its own, and this design does not ask it to: it contributes 8 of 151 rows inside a task-clustered bootstrap that resamples whole tasks. + +## Running the confirmatory arm + +```bash +node --import tsx scripts/tb-gated-stop-ab.ts design # re-seal; prints both digests +node --import tsx scripts/tb-gated-stop-ab.ts confirm # identity gate, then both arms +``` + +`design` writes to the work directory. Copy the result over `benchmarks/trace-repair/gated-stop-ab/design.json` to move the checked-in seal forward; the next `design` reads that file to report the digest it replaces. + +`confirm` runs the control arm at the seat's concurrency limit and the treatment arm one row at a time. +The treatment arm is serial because its pool is sequential state: a row draws the budget that earlier rows returned, so processing order decides allocation. +Running those rows concurrently would change the allocation the seal registered, which makes the scheduling part of the experiment rather than a detail of it. + +The control arm therefore finishes in about a third of the wall time of the treatment arm on the same draw. + +Every finished row is written to `confirm-runs.json` before the next row starts. +A re-invocation reads that file, skips rows already held, and rebuilds the pool from the entitlement and spend of each held treatment row. +An interrupted run costs the row in flight and nothing before it. + +## Reading the result + +`confirm-report.json` carries the digest that produced it, both contrasts, the matched-budget verdict and the served model ids observed across every step. + +The matched-budget rule is a refusal object, not an assertion. +The treatment arm can only spend returned budget on rows that come after the row that returned it, so budget freed by the last rows has nowhere to go. +When that trailing shortfall pushes the arms more than 5 % apart, the registered decision table returns `contrast-refused-unmatched-budget` and the contrast is not read. From 96089e22224bcc85c08a55ee9072fbeb7414d2a9 Mon Sep 17 00:00:00 2001 From: Drew Stone Date: Sat, 15 Aug 2026 03:57:12 -0600 Subject: [PATCH 4/4] fix(trace-repair): write the verdict as a number and refuse spend when the power halt fires The paired-mean-diff estimand reads 'passed' as a number. The confirm report now writes 1 or 0, so the registered contrast computes instead of throwing after both arms have run. Confirm mode now enforces the registered refuse-spend halt before the identity probe. A draw that fails the power floor writes confirm-refusal.json and exits without spend. The study doc records the completed run: 302 rows, contrast +0.0596, 95% CI [-0.0061, +0.1210], verdict no-effect-resolved-at-this-n, with the 429 degradation and the report-time crash as run provenance. --- docs/trace-repair-gated-stop.md | 13 +++++++++++++ scripts/tb-gated-stop-ab.ts | 29 +++++++++++++++++++++++++++-- 2 files changed, 40 insertions(+), 2 deletions(-) diff --git a/docs/trace-repair-gated-stop.md b/docs/trace-repair-gated-stop.md index 5b7b6a2c..f3c38e3c 100644 --- a/docs/trace-repair-gated-stop.md +++ b/docs/trace-repair-gated-stop.md @@ -99,3 +99,16 @@ An interrupted run costs the row in flight and nothing before it. The matched-budget rule is a refusal object, not an assertion. The treatment arm can only spend returned budget on rows that come after the row that returned it, so budget freed by the last rows has nowhere to go. When that trailing shortfall pushes the arms more than 5 % apart, the registered decision table returns `contrast-refused-unmatched-budget` and the contrast is not read. + +## The measured result + +The confirmatory run completed both arms over the sealed draw: 302 rows graded, 151 per arm. + +The corrected primary contrast, best intermediate state in both arms, is **+0.0596** with a 95 % cluster-bootstrap interval of **[-0.0061, +0.1210]**. +The interval includes zero, so the registered decision table reads `no-effect-resolved-at-this-n`. +The study does not certify a gated-stop advantage at this draw. + +Two degradations bound what this run can claim. + +- 15 gated-arm rows were degraded by provider 429 rate-limit responses, and the treatment arm carries all of them; the contrast above is the corrected value after the tail audit accounted for them. +- The registered estimand pipeline as first shipped crashed at report time, because the evidence rows carried a boolean where the paired-mean-diff estimand requires a number; the contrast above was recomputed from the persisted `confirm-runs.json` after the defect was fixed. The runner now writes the verdict as 1 or 0. diff --git a/scripts/tb-gated-stop-ab.ts b/scripts/tb-gated-stop-ab.ts index 2e9da81e..db85dab6 100644 --- a/scripts/tb-gated-stop-ab.ts +++ b/scripts/tb-gated-stop-ab.ts @@ -1225,6 +1225,30 @@ async function main(): Promise { if (mode === 'confirm') { const log = logger(join(WORK, 'confirm.log')) + // The registered halt runs before a token is spent. A draw that fails the + // power floor records `refuse-spend` in design.json; opening spend anyway + // would run an experiment the registration refused. + if (halt.fired) { + const refusal = { + generatedAt: new Date().toISOString(), + sealDigest: sealed.digest, + halt, + powerGate, + onFail: 'refuse-spend', + spentUsd: 0, + rowsGraded: 0, + note: + 'The registered halt fired on the power floor before any row was run. No arm was ' + + 'started and no token was spent. The refusal is the verdict object for this draw.', + } + writeFileSync(join(WORK, 'confirm-refusal.json'), `${JSON.stringify(refusal, null, 2)}\n`) + process.stdout.write( + `REFUSE-SPEND powerFloor: halt fired for gates [${halt.failedGates.join(', ')}]; no spend, no rows graded\n`, + ) + process.exitCode = 4 + return + } + // The identity gate runs before a row is graded and before a token is // spent. A seat that answers the pinned id with another model produces a // contrast for a model nobody registered, so the registered action here is @@ -1337,7 +1361,8 @@ async function main(): Promise { persist() } - // ── The registered contrast. + // ── The registered contrast. The estimand reads `passed` as a number, + // so the boolean verdict is written as 1 or 0 here. const armRows = (arm: string, passedOf: (run: RowRun) => boolean): EvidenceRecord[] => rows .map((row) => held[`${arm}::${row.rowId}`]) @@ -1346,7 +1371,7 @@ async function main(): Promise { rowId: run.rowId, taskName: run.taskName, arm: arm === 'blind-continue' ? 'blind-continue' : 'gated-continue', - passed: passedOf(run), + passed: passedOf(run) ? 1 : 0, })) const finalPassed = (run: RowRun): boolean => run.steps.at(-1)?.gradePassed === true const treatment = armRows('gated-continue', (run) => run.bestPassed)