Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
27 changes: 18 additions & 9 deletions config/publication_protocol.json
Original file line number Diff line number Diff line change
@@ -1,6 +1,7 @@
{
"schema_version": 1,
"frozen_at_utc": "2026-07-15T21:17:38Z",
"last_amended_at_utc": "2026-07-18T21:11:17Z",
"primary_endpoint": "candidate.summary.mean_score under score-v1",
"secondary_endpoints": [
"paired lift versus the full baseline panel",
Expand All @@ -11,14 +12,22 @@
"per-mechanic accepted and rejected outcomes"
],
"output_policy": {
"status": "frozen-fixed-budget",
"basis": "fixed-safety-ceiling",
"output_token_cap": 1024,
"reasoning_enabled": false,
"rationale": "The superseded Luna run recorded 601 API calls with output-token p50 121, p95 210, p99 264, maximum 299, and zero calls at the 1,024-token ceiling. With reasoning disabled and compact JSON actions, a four-cap score sweep would mostly vary a non-binding response limit rather than a defensible inference-compute treatment.",
"pre_full_panel_cap_pressure_rule": "Smoke every registered model at 1,024 before any full-panel result. If any smoke API call emits at least 768 output tokens, or provider or adapter telemetry shows cap-induced truncation, amend the common lane to 2,048 and repeat affected smokes before running any full-panel score. Do not change the cap after full-panel scores are visible.",
"status": "frozen-native-reasoning-cap",
"basis": "common-safety-ceiling-with-native-minimum-reasoning",
"output_token_cap": 4096,
"cap_pressure_threshold_tokens": 3072,
"fallback_output_token_cap": 8192,
"reasoning_policy": "native-minimum",
"rationale": "The user-curated phase-one frontier panel requires native-minimum reasoning: disabled where optional and the lowest supported effort where mandatory. All ten accepted exact-route smokes completed cleanly at a common 4,096-token total-output ceiling, with a maximum per-call output of 1,432 tokens and no truncation or cap-pressure trigger.",
"pre_full_panel_cap_pressure_rule": "Before any full-panel result, smoke every registered model at 4,096 total output tokens. If any smoke API call emits at least 3,072 total output tokens, including reported reasoning tokens, or shows provider or adapter evidence of cap-induced truncation, amend the whole lane to 8,192 and repeat affected smokes before running a full-panel score. Do not change the cap after full-panel scores are visible.",
"efficiency_reporting": "Report actual input tokens, output tokens, cost, and latency per decision as secondary metrics. They do not alter the benchmark score.",
"human_override": "forbidden after any full-panel result is visible; amend this file and the lane config before running an official full-panel cell"
"human_override": "forbidden after any full-panel result is visible; amend this file and the lane config before running an official full-panel cell",
"amendment_2026_07_18": {
"amended_at_utc": "2026-07-18T21:11:17Z",
"evidence_state": "pre-data; no full-panel model result existed",
"superseded_policy": "retired 1,024-token reasoning-disabled safety ceiling with a 768-token trigger and 2,048-token fallback",
"rationale": "Reconcile this pre-registration document with the already frozen and smoke-validated 4,096-token native-minimum-reasoning lane before any full-panel result. This records the existing machine-enforced policy and does not react to panel scores."
}
},
"superseded_output_budget_decision_rule": {
"status": "retired-before-replacement-official-cells",
Expand Down Expand Up @@ -58,14 +67,14 @@
"stop_before_next_cell_when_spend_ceiling_reached": true,
"stop_on_route_or_endpoint_snapshot_drift": true,
"stop_on_missing_cost_or_usage_coverage": true,
"smoke_completion": "10 accepted registered-model smokes at the common cap, with zero cap-pressure or truncation triggers; otherwise amend the whole lane to 2,048 before any full-panel result. Acceptance is machine-checked: each smoke must be recorded via `scripts/run_publication_matrix.py record-smoke` as an accepted entry in config/sota_v2_smoke_manifest.json, and the panel phase refuses to run until every registered model has one.",
"smoke_completion": "10 accepted registered-model smokes at the common 4,096-token cap, with zero 3,072-token cap-pressure or truncation triggers; otherwise amend the whole lane to 8,192 before any full-panel result. Acceptance is machine-checked: each smoke must be recorded via `scripts/run_publication_matrix.py record-smoke` as an accepted entry in config/sota_v2_smoke_manifest.json, and the panel phase refuses to run until every registered model has one.",
"headline_completion": "at least 8 strictly sota-v2 eligible pre-registered rows at the frozen cap"
},
"budget_policy": {
"operator_must_pass_max_spend_usd": true,
"approval_note": "A planning estimate is not authorization to spend. The operator explicitly approves spend by invoking the runner with --max-spend-usd.",
"cost_estimate_artifact": "results/analysis/output-budget-cost-estimate.json",
"cost_estimate_status": "fixed-panel estimate committed; refresh from accepted all-model smoke telemetry before panel approval"
"cost_estimate_status": "fixed-panel estimate refreshed from accepted all-model smoke telemetry; serial reservations include every configured repair attempt plus the committed 1.2x cost contingency"
},
"statistical_analysis_plan": {
"frozen_at_utc": "2026-07-16T01:51:32Z",
Expand Down
2 changes: 2 additions & 0 deletions docs/PUBLISH_READINESS.md
Original file line number Diff line number Diff line change
Expand Up @@ -564,6 +564,7 @@ decision and why.
| 2026-07-17 | Freeze the ten-model phase-one registry and 4,096-token lane after the accepted smoke gate. | All ten registered models completed four decisions with zero failed decisions and zero truncations. Peak per-call output was 1,432 tokens, below the 3,072 cap-pressure trigger. Accepted-route artifact spend was $0.427613; total campaign spend was $0.728909 including the excluded Kimi diagnostic. | Record all ten manifest entries, freeze the registry and native-reasoning cap, retain excluded-model diagnostics, regenerate the cost plan, and unlock panel dry-runs without starting paid panel cells. |
| 2026-07-18 | Settle successful serial-cell reservations against measured spend. | The runner retained every historical worst-case reservation, so a healthy panel could stop against cumulative hypothetical spend even after completed artifacts and the OpenRouter account established a much lower real cost. | Mark successful-cell reservations settled after post-cell spend measurement, keep failed/interrupted reservations active, and evaluate each next cell against measured spend plus only unresolved liabilities. |
| 2026-07-18 | Amend GLM 5.2 from the unhealthy first-party Z.AI FP8 endpoint to Novita FP8. | The frozen `z-ai/fp8` endpoint remained at OpenRouter status `-2` across repeated launch preflights, while the exact dated Novita FP8 endpoint was healthy and advertised the common lane parameters. No full-panel GLM result existed. | Pin `novita/fp8`, replace rather than reuse the Z.AI smoke entry, refresh route pricing/runtime evidence, and require a clean exact-route smoke before restoring panel unlock. The replacement smoke completed 4/4 decisions with zero failures or truncations for $0.009225. |
| 2026-07-18 | Reconcile the frozen publication protocol and reserve repair-call contingency before launch. | Independent Fable 5 review found that the runner and lane correctly enforced 4,096/3,072/8,192 native-minimum reasoning, but `publication_protocol.json` still described the retired 1,024/768/2,048 policy. It also noted that the prior reservation covered only primary calls even though one bounded repair is configured. No full-panel result existed. | Record the current lane as an explicit pre-data protocol amendment. Reserve one full-price call for every configured repair attempt and apply the committed 1.2x cost contingency before admitting each serial cell. Use a sub-$100 operator ceiling and monitor measured spend after every cell. |

## Experiment and release log

Expand All @@ -584,6 +585,7 @@ than pasting large outputs.
| 2026-07-15 | Sweep cost/runtime plan | Superseded by fixed-cap policy | `results/analysis/output-budget-cost-estimate.json` | The prior figures describe the retired 12-cell matrix. Replace with a full-panel estimate after all ten route smokes. |
| 2026-07-16 | First fixed-1,024 smoke series | Superseded by deliberate panel revision | `docs/run_logs/sota-v2-smokes-2026-07-16.md` | Six routes were accepted, two completed with protocol failures, Nemotron Nano exhausted infrastructure retries, and Claude direct remained unhealthy. The evidence remains auditable but cannot unlock the revised 4,096-token native-reasoning panel. |
| 2026-07-15 | Statistical analysis plan | Frozen | `config/publication_protocol.json` | Pre-registered pre-data: unit of inference, primary paired contrast, Holm-Bonferroni multiplicity, descriptive inference labels, tiered ranking, power disclosure, temperature policy, and registry exclusion criteria. |
| 2026-07-18 | Final Fable 5 launch audit | Conditions resolved pre-data | `docs/run_logs/sota-v2-final-launch-audit-2026-07-18.md` | No P0 blocker. Reconciled the stale output-policy text, strengthened reservations for repairs plus contingency, selected a $95 operator ceiling, and retained Tencent timing and per-cell spend monitoring as launch conditions. |

## Living-document maintenance checklist

Expand Down
40 changes: 40 additions & 0 deletions docs/run_logs/sota-v2-final-launch-audit-2026-07-18.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,40 @@
# sota-v2 final launch audit — 2026-07-18

An independent read-only Claude Fable 5 review of merged commit `0f12f21`
returned **GO WITH CONDITIONS** and found no P0 launch blocker. It verified the
ten-model registry, exact-route smoke manifest, GLM Novita amendment, serial
runner, endpoint preflight, resume behavior, status reporting, and publication
locks. Two pre-launch conditions were accepted:

1. reconcile the stale 1,024/768/2,048 policy text in
`config/publication_protocol.json` with the machine-enforced and
smoke-validated 4,096/3,072/8,192 native-minimum-reasoning lane; and
2. account for the configured bounded protocol repair in the per-cell spend
reservation, not only the primary call.

Both conditions were resolved before any full-panel result. The protocol now
records the current lane as a pre-data amendment. The serial runner now reserves
every configured repair attempt as another full-price call and applies the
committed 1.2x cost contingency before a cell may start. Failed or interrupted
reservations remain active; successful cells settle to measured spend.

Using each accepted smoke's measured spend scaled by the panel's 120x decision
ratio, the expected full-panel spend is **$46.7742**. Simulating registry order
with the strengthened reservations produces a maximum expected commitment of
**$89.3659** immediately before Mistral. A **$95 operator ceiling** therefore
keeps authorization below the user's $100 limit while leaving approximately
$5.63 above that conservative expected commitment.

The ceiling is a cell-boundary guard, not a provider-side billing limit. The
reservation assumes 8,000 input tokens per decision, the frozen 4,096-token
output cap, every configured repair call, and 1.2x contingency. The operator
must monitor measured spend after every cell and stop on unexpected divergence.

Final operational conditions:

- use a fresh panel run directory and one serial runner process;
- confirm `GM_BENCH_PRIVATE_SEEDS` is unset;
- avoid unrelated OpenRouter usage during account-delta measurement;
- start promptly and complete the free Tencent HY3 cell before its July 21
catalog expiration; and
- keep the status watcher open and review measured spend after each cell.
22 changes: 21 additions & 1 deletion scripts/run_publication_matrix.py
Original file line number Diff line number Diff line change
Expand Up @@ -403,9 +403,17 @@ def _cell_reservation_usd(cell: Cell) -> float:
preset = PRESETS[cell.preset]
decisions = len(preset["seeds"]) * int(preset["seasons"]) * len(PHASES) * cell.repeats
input_tokens = int(assumptions["input_tokens_per_decision"])
repair_attempts = int(cell.fixed_options.get("GM_BENCH_PROTOCOL_REPAIR_ATTEMPTS", "0"))
contingency = float(assumptions["cost_contingency_multiplier"])
if input_tokens < 1 or repair_attempts < 0 or contingency < 1:
raise ValueError("publication reservation assumptions must be positive and conservative")
prompt = decisions * input_tokens * float(rates["prompt"])
completion = decisions * cell.cap * float(rates["completion"])
return round(prompt + completion, 6)
# Reserve every configured repair as another full-price call, then apply
# the committed planning contingency. The guard still acts at cell
# boundaries, so this deliberately overstates the likely liability before
# a cell is allowed to start.
return round((prompt + completion) * (1 + repair_attempts) * contingency, 6)


def _write_json_atomic(path: Path, payload: dict[str, Any]) -> None:
Expand Down Expand Up @@ -927,6 +935,12 @@ def _reserve_cell(run_dir: Path, cell: Cell, measured_spend: float, ceiling: flo
stored["total_reserved_usd"] = float(stored.get("total_reserved_usd") or 0) + reservation
stored["attempts"] = int(stored.get("attempts") or 1) + 1
stored["status"] = "active"
stored["protocol_repair_attempts_reserved_per_decision"] = int(
cell.fixed_options.get("GM_BENCH_PROTOCOL_REPAIR_ATTEMPTS", "0")
)
stored["cost_contingency_multiplier"] = float(
_read_json(PRICING_CONFIG)["planning_assumptions"]["cost_contingency_multiplier"]
)
stored.pop("settled_at_utc", None)
stored.pop("measured_run_spend_usd", None)
_write_json_atomic(path, payload)
Expand All @@ -948,6 +962,12 @@ def _reserve_cell(run_dir: Path, cell: Cell, measured_spend: float, ceiling: flo
"total_reserved_usd": reservation,
"attempts": 1,
"status": "active",
"protocol_repair_attempts_reserved_per_decision": int(
cell.fixed_options.get("GM_BENCH_PROTOCOL_REPAIR_ATTEMPTS", "0")
),
"cost_contingency_multiplier": float(
_read_json(PRICING_CONFIG)["planning_assumptions"]["cost_contingency_multiplier"]
),
}
_write_json_atomic(path, payload)
print(f"reserved ${reservation:.4f} for {stem}; cumulative conservative commitment ${committed + reservation:.4f}")
Expand Down
9 changes: 9 additions & 0 deletions tests/test_publication.py
Original file line number Diff line number Diff line change
Expand Up @@ -228,6 +228,7 @@ def test_publication_model_registry_is_consistent_with_revised_lane() -> None:
sweep = json.loads(Path("config/output_budget_sweep.json").read_text())
registry = json.loads(Path("config/sota_v2_models.json").read_text())
lane = json.loads(Path("config/sota_v2_lane.json").read_text())
protocol = json.loads(Path("config/publication_protocol.json").read_text())

models = registry["models"]
identities = {(row["provider"], row["model"]): row for row in models}
Expand Down Expand Up @@ -258,6 +259,14 @@ def test_publication_model_registry_is_consistent_with_revised_lane() -> None:
assert lane["model_registry"] == "config/sota_v2_models.json"
assert lane["minimum_headline_models"] >= 8

output_policy = protocol["output_policy"]
assert output_policy["status"] == lane["output_budget_status"]
assert output_policy["basis"] == lane["output_policy_basis"]
assert output_policy["reasoning_policy"] == lane["reasoning_policy"]
assert output_policy["output_token_cap"] == lane["output_token_cap"]
assert output_policy["cap_pressure_threshold_tokens"] == lane["cap_pressure_threshold_tokens"]
assert output_policy["fallback_output_token_cap"] == lane["fallback_output_token_cap"]

glm = identities[("openrouter", "z-ai/glm-5.2")]
assert glm["id"] == "openrouter-glm-5.2-novita"
assert glm["upstream_provider"] == "Novita"
Expand Down
12 changes: 12 additions & 0 deletions tests/test_publication_runner.py
Original file line number Diff line number Diff line change
Expand Up @@ -656,6 +656,18 @@ def test_cell_reservation_blocks_launch_before_ceiling_overrun(tmp_path: Path) -
assert not (tmp_path / "openrouter-reservations.json").exists()


def test_cell_reservation_covers_repairs_and_cost_contingency() -> None:
cell = build_cells("panel", model_id="openrouter-gpt-5.6-luna-openai")[0]
pricing = json.loads(Path("config/openrouter_pricing_snapshot.json").read_text())
assumptions = pricing["planning_assumptions"]
rates = pricing["models"][cell.model]
decisions = 8 * 5 * 4 * 3
base = decisions * (assumptions["input_tokens_per_decision"] * rates["prompt"] + cell.cap * rates["completion"])

assert cell.fixed_options["GM_BENCH_PROTOCOL_REPAIR_ATTEMPTS"] == "1"
assert _cell_reservation_usd(cell) == pytest.approx(base * 2 * assumptions["cost_contingency_multiplier"], abs=1e-6)


def test_retry_reservation_accounts_for_fresh_full_attempt(tmp_path: Path) -> None:
cell = build_cells("smoke", model_id="openrouter-gpt-5.6-luna-openai", cap=4096)[0]
reservation = _cell_reservation_usd(cell)
Expand Down