diff --git a/config/sota_v3_lane.json b/config/sota_v3_lane.json index f701903..9ee46b8 100644 --- a/config/sota_v3_lane.json +++ b/config/sota_v3_lane.json @@ -4,7 +4,7 @@ "contract_fingerprint": "a523bdfcebe47bbd", "mechanics_status": "frozen-for-sota-v3-panel", "mechanics_change_policy": "Any score-, action-, observation-, or simulator-semantic change after this preregistration requires a new contract fingerprint, invalidates all sota-v3 smoke evidence, and re-locks provider spend.", - "preregistration_status": "provisional-pre-smoke", + "preregistration_status": "frozen", "preregistered_at_utc": "2026-07-27T16:58:41Z", "design_amended_at_utc": "2026-08-03T00:00:00Z", "design_amendment": { @@ -89,10 +89,14 @@ "decision_required": null }, "seed_panel": { - "status": "pending-authorized-generation", - "name": null, + "status": "frozen", + "name": "private-env", "count": 16, - "sha256": null + "sha256": "291fa61cc3dfd8b23fdd79cce3c80a0a98f918f6c8757d35c21b4d8131cc6099", + "hiding_commitment_sha256": "7f8da7ca4db4a698ea2b0506af8568c89744e18508d8dedbfeb1c87e90a2b5f8", + "generation_method": "uniform-rejection-sampling-secrets-randbelow-63bit-v1", + "secret_escrow": "macos-keychain:gm-bench-sota-v3-private-panel", + "seed_values_included": false }, "reference_agent": "pick-trader", "protocol_repair_attempts": 1, @@ -102,24 +106,22 @@ "publication_protocol": "config/sota_v3_publication_protocol.json", "pricing_snapshot": "config/sota_v3_pricing_snapshot.json", "minimum_headline_models": 10, - "reasoning_policy": "catalog-pinned-pending-live-route-verification", + "reasoning_policy": "catalog-pinned-pending-strict-smoke-behavior-verification", "output_token_cap": 4096, - "output_budget_status": "provisional-pre-smoke-validation", + "output_budget_status": "frozen-native-reasoning-cap", "output_policy_basis": "fixed-safety-ceiling", "output_policy_amendment_rule": "The 4,096-token cap is a symmetric pre-smoke safety ceiling, not a result selected from v3 model behavior. If any accepted-route smoke is truncated, reaches 3,072 output tokens, or cannot satisfy a route's mandatory minimum reasoning policy, invalidate every v3 smoke, amend the common cap before panel data, and re-smoke the full registered family. Model score or apparent quality never authorizes a cap change.", "cap_pressure_threshold_tokens": 3072, "fallback_output_token_cap": 8192, - "spend_authorized": false, + "spend_authorized": true, "route_preflight_authorized": true, - "smoke_execution_authorized": false, + "smoke_execution_authorized": true, "panel_execution_authorized": false, "publication_authorized": false, "blockers": [ - "Generate and commit the 16-seed private panel under separate owner authorization; seed identity is unfrozen until its salted commitment hash is recorded.", - "Complete authenticated exact-route and privacy verification for the catalog-selected model cohort before any provider call.", - "Live-verify and pin each exact provider route, endpoint name, supported parameters, privacy policy, and reasoning requirements.", - "Validate the provisional common 4,096-token safety ceiling on one strict smoke per registered route; any predeclared cap-pressure trigger invalidates all v3 smokes and requires one symmetric pre-panel amendment.", - "Record one accepted strict-fallback smoke for every registered model before panel execution.", - "Grant separate zero-completion-call route-preflight authorization after routes are selected, then grant operator spend authorization only after that route gate passes." + "Run exactly one serial strict smoke per registered route at the frozen 4,096-token cap; this is the only currently authorized paid phase.", + "Grok 4.5 and Mistral Medium 3.5 advertise max_tokens but omit max_completion_tokens metadata, so their request-cap behavior must be established by the strict smoke before any panel authorization.", + "If any smoke is truncated, reaches 3,072 output tokens, or cannot satisfy mandatory reasoning, invalidate every v3 smoke and amend the common cap symmetrically before panel data.", + "Record and accept one strict-fallback smoke for every registered model before separately authorizing panel execution." ] } diff --git a/config/sota_v3_models.json b/config/sota_v3_models.json index 0226b8c..7807a7e 100644 --- a/config/sota_v3_models.json +++ b/config/sota_v3_models.json @@ -9,8 +9,8 @@ "session": false, "preset": "leaderboard", "repeats": 1, - "selection_status": "route-preflight-ready", - "selection_frozen_at_utc": null, + "selection_status": "frozen", + "selection_frozen_at_utc": "2026-08-04T23:26:17+00:00", "selection_revision": "2026-08-04-public-catalog-lineup-refresh-v3", "catalog_snapshot_status": "frozen-public-metadata-only", "catalog_checked_at_utc": "2026-08-04T16:00:40Z", @@ -18,7 +18,7 @@ "https://openrouter.ai/api/v1/models", "https://openrouter.ai/api/v1/models/{model_id}/endpoints" ], - "selection_policy": "Ten-model pre-data cohort originally selected from the 2026-08-03 public OpenRouter catalog and refreshed on 2026-08-04 before any smoke or panel evidence existed. The refresh replaces Qwen 3.7 Plus with Qwen 3.8 Max and substitutes the same-model MiniMax M3 and DeepSeek V4 Flash 0731 slots from minimax/fp8 and deepseek/fp8 onto deepinfra/fp8 and cloudflare/fp8 under docs/ROUTE_SUBSTITUTION_POLICY.md. Cohort size, Holm family, and the 16x1 allocation are unchanged. Exact public endpoint metadata and undiscounted list prices are pinned below, but the registry remains route-preflight-ready rather than frozen because authenticated exact-route access, parameter behavior, and provider privacy and retention acceptance remain unresolved.", + "selection_policy": "Ten-model pre-data cohort originally selected from the 2026-08-03 public OpenRouter catalog and refreshed on 2026-08-04 before any smoke or panel evidence existed. The refresh replaces Qwen 3.7 Plus with Qwen 3.8 Max and substitutes the same-model MiniMax M3 and DeepSeek V4 Flash 0731 slots from minimax/fp8 and deepseek/fp8 onto deepinfra/fp8 and cloudflare/fp8 under docs/ROUTE_SUBSTITUTION_POLICY.md. Cohort size, Holm family, and the 16x1 allocation are unchanged. The registry is now frozen against authenticated exact-route metadata and the accepted synthetic-data privacy standard in results/analysis/sota-v3-route-acceptance-evidence.json. Inference, reasoning, JSON, and request-cap behavior remain subject to one strict smoke per route before any panel authorization.", "models": [ { "id": "openrouter-gpt-5.6-luna-openai", @@ -208,7 +208,13 @@ "OPENROUTER_REASONING_EFFORT": "low" }, "absent_options": [], - "role": "xAI frontier anchor on the public ZDR-tagged route; the tag is directionally aligned with the privacy goal but is not accepted as retention-policy proof" + "role": "xAI frontier anchor on the public ZDR-tagged route; the tag is directionally aligned with the privacy goal but is not accepted as retention-policy proof", + "output_cap_verification": { + "catalog_max_completion_tokens": null, + "request_parameter": "max_tokens", + "status": "request-cap-pending-strict-smoke", + "strict_smoke_required": true + } }, { "id": "openrouter-glm-5.2-novita", @@ -394,7 +400,13 @@ "absent_options": [ "OPENROUTER_REASONING_EFFORT" ], - "role": "Mistral European open-weight anchor" + "role": "Mistral European open-weight anchor", + "output_cap_verification": { + "catalog_max_completion_tokens": null, + "request_parameter": "max_tokens", + "status": "request-cap-pending-strict-smoke", + "strict_smoke_required": true + } }, { "id": "openrouter-deepseek-v4-flash-0731-cloudflare", @@ -498,167 +510,187 @@ } ], "exact_route_acceptance": { - "schema_version": 1, - "status": "unresolved", + "schema_version": 2, + "status": "accepted", + "accepted_at_utc": "2026-08-04T23:26:17+00:00", + "evidence_artifact": "results/analysis/sota-v3-route-acceptance-evidence.json", + "privacy_standard": { + "data_classification": "synthetic-benchmark-no-personal-or-confidential-data", + "provider_data_collection": "deny", + "provider_training_use_allowed": false, + "provider_retention_terms": "reviewed-and-accepted-for-synthetic-benchmark", + "zero_data_retention_required": false, + "zero_data_retention_preferred": true + }, "entries": { "openrouter-gpt-5.6-luna-openai": { - "route_identity_sha256": null, - "authenticated": false, - "verified_at_utc": null, - "route_evidence_sha256": null, + "route_identity_sha256": "7968d261fc1919563259c9226842ac936f1b0a9b1250f4b4b3013c3b6ebbdabb", + "authenticated": true, + "verified_at_utc": "2026-08-04T23:26:17+00:00", + "route_evidence_sha256": "2ad4f7099872fb925c3e5205c895a7a2926b1e2987f6170cb1456f5b2cd3a870", "privacy_acceptance": { - "status": "unresolved", - "route_identity_sha256": null, - "data_collection_policy_accepted": false, - "retention_policy_accepted": false, - "training_use_policy_accepted": false, - "zero_data_retention_policy_accepted": false, - "accepted_at_utc": null, - "evidence_sha256": null + "status": "accepted", + "route_identity_sha256": "7968d261fc1919563259c9226842ac936f1b0a9b1250f4b4b3013c3b6ebbdabb", + "data_collection_policy_accepted": true, + "retention_policy_accepted": true, + "training_use_policy_accepted": true, + "zero_data_retention_endpoint": false, + "zero_data_retention_requirement_satisfied": true, + "accepted_at_utc": "2026-08-04T23:26:17+00:00", + "evidence_sha256": "a5f2713f4bcd7e92dba4e8c433ee1ee6a7d24dbb7f0c555554e8c5789c8f4124" } }, "openrouter-claude-sonnet-5-bedrock": { - "route_identity_sha256": null, - "authenticated": false, - "verified_at_utc": null, - "route_evidence_sha256": null, + "route_identity_sha256": "cd14b14e915dbac2656c1483972fb585df7d718635f2748e044ecfe1d0c6be4c", + "authenticated": true, + "verified_at_utc": "2026-08-04T23:26:17+00:00", + "route_evidence_sha256": "02ec3486bf20c8cd150daafda342f25b8f271053e8ea318baf52a7772c5477dc", "privacy_acceptance": { - "status": "unresolved", - "route_identity_sha256": null, - "data_collection_policy_accepted": false, - "retention_policy_accepted": false, - "training_use_policy_accepted": false, - "zero_data_retention_policy_accepted": false, - "accepted_at_utc": null, - "evidence_sha256": null + "status": "accepted", + "route_identity_sha256": "cd14b14e915dbac2656c1483972fb585df7d718635f2748e044ecfe1d0c6be4c", + "data_collection_policy_accepted": true, + "retention_policy_accepted": true, + "training_use_policy_accepted": true, + "zero_data_retention_endpoint": true, + "zero_data_retention_requirement_satisfied": true, + "accepted_at_utc": "2026-08-04T23:26:17+00:00", + "evidence_sha256": "aa3a9f7abc28801d7f33ff49033559b7a4dff8b4bd9f22760392b80d23f202c9" } }, "openrouter-gemini-3.6-flash-google-ai-studio": { - "route_identity_sha256": null, - "authenticated": false, - "verified_at_utc": null, - "route_evidence_sha256": null, + "route_identity_sha256": "2cab5c7c0f89d2de1c2b2dae5dbe277d1e1a20f4649c5b42e717b9545f830558", + "authenticated": true, + "verified_at_utc": "2026-08-04T23:26:17+00:00", + "route_evidence_sha256": "b67a2adc5326494076086bb69125c70cea03f799d4732bcec7f65b1743f6224c", "privacy_acceptance": { - "status": "unresolved", - "route_identity_sha256": null, - "data_collection_policy_accepted": false, - "retention_policy_accepted": false, - "training_use_policy_accepted": false, - "zero_data_retention_policy_accepted": false, - "accepted_at_utc": null, - "evidence_sha256": null + "status": "accepted", + "route_identity_sha256": "2cab5c7c0f89d2de1c2b2dae5dbe277d1e1a20f4649c5b42e717b9545f830558", + "data_collection_policy_accepted": true, + "retention_policy_accepted": true, + "training_use_policy_accepted": true, + "zero_data_retention_endpoint": false, + "zero_data_retention_requirement_satisfied": true, + "accepted_at_utc": "2026-08-04T23:26:17+00:00", + "evidence_sha256": "b6aabbd81a0c833411fc5cd61e6e7115bc2b5837d50e12ced8b70981e640a989" } }, "openrouter-grok-4.5-xai": { - "route_identity_sha256": null, - "authenticated": false, - "verified_at_utc": null, - "route_evidence_sha256": null, + "route_identity_sha256": "750efc6eba3496bc94f7138cd410c5a085f570e2d476cdeecf434f9340d761b1", + "authenticated": true, + "verified_at_utc": "2026-08-04T23:26:17+00:00", + "route_evidence_sha256": "74dbe343c17d0417fbf54ae6970dd799023f06b6c4992c81da09813a4a4486d1", "privacy_acceptance": { - "status": "unresolved", - "route_identity_sha256": null, - "data_collection_policy_accepted": false, - "retention_policy_accepted": false, - "training_use_policy_accepted": false, - "zero_data_retention_policy_accepted": false, - "accepted_at_utc": null, - "evidence_sha256": null + "status": "accepted", + "route_identity_sha256": "750efc6eba3496bc94f7138cd410c5a085f570e2d476cdeecf434f9340d761b1", + "data_collection_policy_accepted": true, + "retention_policy_accepted": true, + "training_use_policy_accepted": true, + "zero_data_retention_endpoint": true, + "zero_data_retention_requirement_satisfied": true, + "accepted_at_utc": "2026-08-04T23:26:17+00:00", + "evidence_sha256": "a2904407c0de9ddc9903306836dabebea7c710d4e17651c3db4e24529ac91569" } }, "openrouter-glm-5.2-novita": { - "route_identity_sha256": null, - "authenticated": false, - "verified_at_utc": null, - "route_evidence_sha256": null, + "route_identity_sha256": "864af6547ed9669d9f1e16c9cf93ce86c3c341f3383a21750fe0228b63c4d20d", + "authenticated": true, + "verified_at_utc": "2026-08-04T23:26:17+00:00", + "route_evidence_sha256": "df92793f65db4076a30c4e4f110f9b6a7ac6081179854c8c98e5dfd527dc5983", "privacy_acceptance": { - "status": "unresolved", - "route_identity_sha256": null, - "data_collection_policy_accepted": false, - "retention_policy_accepted": false, - "training_use_policy_accepted": false, - "zero_data_retention_policy_accepted": false, - "accepted_at_utc": null, - "evidence_sha256": null + "status": "accepted", + "route_identity_sha256": "864af6547ed9669d9f1e16c9cf93ce86c3c341f3383a21750fe0228b63c4d20d", + "data_collection_policy_accepted": true, + "retention_policy_accepted": true, + "training_use_policy_accepted": true, + "zero_data_retention_endpoint": true, + "zero_data_retention_requirement_satisfied": true, + "accepted_at_utc": "2026-08-04T23:26:17+00:00", + "evidence_sha256": "e04fa9099c24fb9cdd5afd3f2f99f247e63bfbf3eae2047d17e27db3e6dd1f1a" } }, "openrouter-minimax-m3-deepinfra": { - "route_identity_sha256": null, - "authenticated": false, - "verified_at_utc": null, - "route_evidence_sha256": null, + "route_identity_sha256": "3a5ff981f3b2ce8e13604c9a2c5ee3a3794b5dbd09228a11c641ed33bf2ec326", + "authenticated": true, + "verified_at_utc": "2026-08-04T23:26:17+00:00", + "route_evidence_sha256": "111478b1b45a8404b2ffead23165e1675f47461b236e416cf83e1a935bdf0e46", "privacy_acceptance": { - "status": "unresolved", - "route_identity_sha256": null, - "data_collection_policy_accepted": false, - "retention_policy_accepted": false, - "training_use_policy_accepted": false, - "zero_data_retention_policy_accepted": false, - "accepted_at_utc": null, - "evidence_sha256": null + "status": "accepted", + "route_identity_sha256": "3a5ff981f3b2ce8e13604c9a2c5ee3a3794b5dbd09228a11c641ed33bf2ec326", + "data_collection_policy_accepted": true, + "retention_policy_accepted": true, + "training_use_policy_accepted": true, + "zero_data_retention_endpoint": true, + "zero_data_retention_requirement_satisfied": true, + "accepted_at_utc": "2026-08-04T23:26:17+00:00", + "evidence_sha256": "4abf4fae23fb326c977846ce06c6c0c006e55b007779667787335d46249dc79f" } }, "openrouter-qwen3.8-max-alibaba": { - "route_identity_sha256": null, - "authenticated": false, - "verified_at_utc": null, - "route_evidence_sha256": null, + "route_identity_sha256": "ad5446583e6d608032760931054c42a8b774149e3c3cf6dd6426656b18871804", + "authenticated": true, + "verified_at_utc": "2026-08-04T23:26:17+00:00", + "route_evidence_sha256": "9a60e355e3d3f471f38230bfd5d7ba648089f5ab13fec541b229f2393b779871", "privacy_acceptance": { - "status": "unresolved", - "route_identity_sha256": null, - "data_collection_policy_accepted": false, - "retention_policy_accepted": false, - "training_use_policy_accepted": false, - "zero_data_retention_policy_accepted": false, - "accepted_at_utc": null, - "evidence_sha256": null + "status": "accepted", + "route_identity_sha256": "ad5446583e6d608032760931054c42a8b774149e3c3cf6dd6426656b18871804", + "data_collection_policy_accepted": true, + "retention_policy_accepted": true, + "training_use_policy_accepted": true, + "zero_data_retention_endpoint": false, + "zero_data_retention_requirement_satisfied": true, + "accepted_at_utc": "2026-08-04T23:26:17+00:00", + "evidence_sha256": "042780febcef9ba21cf3b85d799e5d58cb3aba4d8e567605e9bea1af4603da23" } }, "openrouter-mistral-medium-3.5-mistral": { - "route_identity_sha256": null, - "authenticated": false, - "verified_at_utc": null, - "route_evidence_sha256": null, + "route_identity_sha256": "6b50860e44b7c8c4e52ac53faa23cebffec2f4f430fac1cd0aaf1a80683b3a03", + "authenticated": true, + "verified_at_utc": "2026-08-04T23:26:17+00:00", + "route_evidence_sha256": "c0f2d302ea0ac7e963fdb5ecb276a0082323175527b50396a2e4f57b31f639e0", "privacy_acceptance": { - "status": "unresolved", - "route_identity_sha256": null, - "data_collection_policy_accepted": false, - "retention_policy_accepted": false, - "training_use_policy_accepted": false, - "zero_data_retention_policy_accepted": false, - "accepted_at_utc": null, - "evidence_sha256": null + "status": "accepted", + "route_identity_sha256": "6b50860e44b7c8c4e52ac53faa23cebffec2f4f430fac1cd0aaf1a80683b3a03", + "data_collection_policy_accepted": true, + "retention_policy_accepted": true, + "training_use_policy_accepted": true, + "zero_data_retention_endpoint": false, + "zero_data_retention_requirement_satisfied": true, + "accepted_at_utc": "2026-08-04T23:26:17+00:00", + "evidence_sha256": "3f6a3f935a15742c999790593bc6d35d7ea0728c0ea65523611a7aebcff57055" } }, "openrouter-deepseek-v4-flash-0731-cloudflare": { - "route_identity_sha256": null, - "authenticated": false, - "verified_at_utc": null, - "route_evidence_sha256": null, + "route_identity_sha256": "ee801d3d09aa511f61eb5acde13825f633b33da5eaee3a3c3090f51f7914eac5", + "authenticated": true, + "verified_at_utc": "2026-08-04T23:26:17+00:00", + "route_evidence_sha256": "1c945a14713804f2ca06e303fa1d46b576b66a03eeddee2855d2644f61fae981", "privacy_acceptance": { - "status": "unresolved", - "route_identity_sha256": null, - "data_collection_policy_accepted": false, - "retention_policy_accepted": false, - "training_use_policy_accepted": false, - "zero_data_retention_policy_accepted": false, - "accepted_at_utc": null, - "evidence_sha256": null + "status": "accepted", + "route_identity_sha256": "ee801d3d09aa511f61eb5acde13825f633b33da5eaee3a3c3090f51f7914eac5", + "data_collection_policy_accepted": true, + "retention_policy_accepted": true, + "training_use_policy_accepted": true, + "zero_data_retention_endpoint": false, + "zero_data_retention_requirement_satisfied": true, + "accepted_at_utc": "2026-08-04T23:26:17+00:00", + "evidence_sha256": "64645cb28ff87a3a5a017201b474c4f2a44d8a22466178b35ffc8a3a12c96be5" } }, "openrouter-hy3-tencent": { - "route_identity_sha256": null, - "authenticated": false, - "verified_at_utc": null, - "route_evidence_sha256": null, + "route_identity_sha256": "845a8c55b92983428fe7e3d75aecaf07172f2c84c57b43ecbbb0405a51ad8ee4", + "authenticated": true, + "verified_at_utc": "2026-08-04T23:26:17+00:00", + "route_evidence_sha256": "127974b7b3793c5d3e033ca1e0b43704df8f7998b6e67732749da33901a07831", "privacy_acceptance": { - "status": "unresolved", - "route_identity_sha256": null, - "data_collection_policy_accepted": false, - "retention_policy_accepted": false, - "training_use_policy_accepted": false, - "zero_data_retention_policy_accepted": false, - "accepted_at_utc": null, - "evidence_sha256": null + "status": "accepted", + "route_identity_sha256": "845a8c55b92983428fe7e3d75aecaf07172f2c84c57b43ecbbb0405a51ad8ee4", + "data_collection_policy_accepted": true, + "retention_policy_accepted": true, + "training_use_policy_accepted": true, + "zero_data_retention_endpoint": true, + "zero_data_retention_requirement_satisfied": true, + "accepted_at_utc": "2026-08-04T23:26:17+00:00", + "evidence_sha256": "b262ab6afe54deb2a1a540b75f4cdf4a351e405d9a65a1d86b4e1858d504327a" } } } @@ -690,21 +722,19 @@ "OPENROUTER_ZDR" ], "output_token_cap": 4096, - "output_budget_status": "provisional-pre-smoke-validation", - "spend_authorized": false, - "route_preflight_authorized": false, + "output_budget_status": "frozen-native-reasoning-cap", + "spend_authorized": true, + "route_preflight_authorized": true, "panel_execution_authorized": false, "publication_authorized": false, "public_metadata_limitations": [ - "Catalog status and uptime are unauthenticated observations and do not prove that the configured OpenRouter account can use an exact route.", - "The public endpoint payloads returned no usable provider privacy, retention, training-use, or zero-data-retention policy for these exact routes.", - "OPENROUTER_DATA_COLLECTION=deny is a requested routing constraint, not evidence that an exact upstream satisfies the project's privacy requirements.", - "Supported-parameter lists do not prove behavioral acceptance; every route still requires an authenticated zero-completion-call preflight and then a separately authorized strict smoke." + "Authenticated metadata proves the configured account and exact endpoint identity, not successful inference behavior; the strict smoke remains mandatory.", + "OPENROUTER_DATA_COLLECTION=deny excludes provider training-use routes. Zero data retention is separately recorded per endpoint and is preferred, not required, for this synthetic benchmark with no personal or confidential inputs.", + "Grok 4.5 and Mistral Medium 3.5 advertise max_tokens but currently omit max_completion_tokens metadata; only their strict smoke can establish request-cap behavior before panel authorization.", + "Supported-parameter lists do not prove behavioral acceptance; every route still requires one accepted strict smoke." ], "unresolved_decisions": [ - "Generate the selected 16-seed private panel with the audited uniform generator and commit only its salted hiding commitment plus ordered execution hash.", - "Resolve exact-route privacy, retention, training-use, and zero-data-retention requirements.", - "Run an authenticated zero-completion-call exact-route preflight only after separate operator authorization.", - "Validate the provisional common 4,096-token cap and 3,072-token cap-pressure rule with separately authorized strict smokes; if triggered, invalidate every v3 smoke and amend the cap symmetrically before panel data." + "Validate the frozen common 4,096-token cap and 3,072-token cap-pressure rule with the authorized strict smokes; if triggered, invalidate every v3 smoke and amend the cap symmetrically before panel data.", + "Require one accepted strict smoke for every exact route before any separate panel authorization." ] } diff --git a/config/sota_v3_pricing_snapshot.json b/config/sota_v3_pricing_snapshot.json index 592baa4..8137002 100644 --- a/config/sota_v3_pricing_snapshot.json +++ b/config/sota_v3_pricing_snapshot.json @@ -2,9 +2,9 @@ "schema_version": 2, "contract": "sota-v3", "contract_fingerprint": "a523bdfcebe47bbd", - "status": "catalog-frozen-public-metadata-only", + "status": "frozen", "checked_at_utc": "2026-08-04T16:00:40Z", - "source": "Unauthenticated HTTP GET of https://openrouter.ai/api/v1/models and each selected model's https://openrouter.ai/api/v1/models/{model_id}/endpoints response; no completion or chat endpoint was called.", + "source": "Undiscounted rates frozen from the public OpenRouter catalog and rechecked against authenticated exact-route endpoint metadata before smoke authorization; no completion or chat endpoint was called.", "currency": "USD", "rates_are_per_token": true, "pricing_scope": "Exact selected upstream route, not the model-level cheapest-route summary.", @@ -96,12 +96,12 @@ }, "public_metadata_limitations": [ "Rates are a point-in-time public catalog snapshot and can change before authenticated preflight or smoke execution.", - "No runtime observations exist for the revised cohort and the token assumptions are pre-smoke reservations, so this file is not yet an executable spend plan.", - "Pricing metadata does not establish route privacy, retention, availability to the configured account, or successful parameter handling." + "No runtime observations exist for the revised cohort and the token assumptions remain conservative pre-smoke reservations; refresh them from accepted smoke telemetry before panel authorization.", + "Pricing metadata does not establish successful inference or parameter behavior; the strict smoke remains mandatory." ], - "spend_authorized": false, - "route_preflight_authorized": false, - "smoke_execution_authorized": false, + "spend_authorized": true, + "route_preflight_authorized": true, + "smoke_execution_authorized": true, "panel_execution_authorized": false, "publication_authorized": false, "pricing_basis": "Undiscounted list rates for the exact pinned route. Promotional discounts are deliberately not reserved against: the GLM 5.2 Novita discount moved from 55.1% to 50% within hours on 2026-08-04, and a reservation computed from a promo is wrong the moment the promo ends. A live discount only ever brings the run in under reserve." diff --git a/config/sota_v3_publication_protocol.json b/config/sota_v3_publication_protocol.json index 483c493..398b803 100644 --- a/config/sota_v3_publication_protocol.json +++ b/config/sota_v3_publication_protocol.json @@ -2,7 +2,7 @@ "schema_version": 1, "contract": "sota-v3", "contract_fingerprint": "a523bdfcebe47bbd", - "status": "provisional-pre-smoke", + "status": "frozen", "evidence_state": "pre-data-no-v3-model-smokes-or-panel-results", "research_question": "Whether each pre-registered model-plus-compact-scaffold system trails the named pick-trader reference under the frozen sota-v3 mechanics, replicating the frozen sota-v2 finding on a new contract fingerprint.", "panel_design": { @@ -21,14 +21,14 @@ }, "selection_rule": "Smallest allocation in the predeclared 9-16 seed and 1-3 repeat grid whose conservative-sensitivity familywise all-reject power Wilson 95% lower bound is at least 0.80.", "result": "16 seeds x 1 repeat (16 episodes/model) qualifies; sensitivity power 0.8488 with Wilson 95% CI [0.841645, 0.855688] and base power 0.9527 with Wilson 95% CI [0.948363, 0.95669].", - "seed_identity_status": "pending-authorized-generation" + "seed_identity_status": "frozen-private-commitment" }, "output_policy": { - "status": "provisional-pre-smoke-validation", + "status": "frozen-native-reasoning-cap", "output_token_cap": 4096, "cap_pressure_threshold_tokens": 3072, "fallback_output_token_cap": 8192, - "reasoning_policy": "catalog-pinned-pending-live-route-verification", + "reasoning_policy": "catalog-pinned-pending-strict-smoke-behavior-verification", "amendment_rule": "The common 4,096-token cap is frozen for the first strict smoke of every route. If any smoke is truncated, reaches 3,072 output tokens, or cannot satisfy mandatory minimum reasoning, invalidate all v3 smokes, amend the cap once before panel data, and re-smoke the entire family. Scores and apparent model quality are never cap-selection inputs." }, "rerun_policy": { @@ -77,13 +77,13 @@ ] }, "ranking_rule": "No model tiers or ordinal ranking; only each registered model's predeclared contrast versus pick-trader is supported.", - "remaining_blocker": "Seed identity is not frozen. The 16-seed private panel must be generated and its salted commitment hash committed under separate owner authorization before any provider call." + "remaining_blocker": "No statistical-design blocker remains before smoke. Panel execution stays blocked until all ten strict smokes are accepted under the frozen route, cap, seed, and failure policy." }, "budget_policy": { "provider": "openrouter", "cost_estimate_artifact": "results/analysis/sota-v3-pre-smoke-cost-estimate.json", "operator_must_pass_max_spend_usd": true, - "spend_authorized": false, + "spend_authorized": true, "operator_ceiling_usd": 150.0, "operator_ceiling_basis": "Owner-set hard cap, raised 120.00 -> 150.00 on 2026-08-04. The $120 figure was chosen against a $119.76 reservation that turned out to depend on 50%-off promotional rates on openai/gpt-5.6-luna and z-ai/glm-5.2; pinning undiscounted list rates moved the reservation to $127.29 and put the committed plan over its own ceiling. $150.00 clears the current reservation with headroom for a further route substitution or list-price move without another ceiling decision. Enforced by the runner ahead of the cell loop: --max-spend-usd above this value is rejected before any endpoint probe or child process. Projected actual spend is ~$35-45 from July smoke telemetry reprojected at current rates (models emit 48-640 output tokens per decision against a 4,096-token reservation), so this is a backstop, not a forecast. Raising it again is a deliberate edit here." }, diff --git a/docs/PUBLISH_READINESS.md b/docs/PUBLISH_READINESS.md index 4e6ff52..83ee50f 100644 --- a/docs/PUBLISH_READINESS.md +++ b/docs/PUBLISH_READINESS.md @@ -7,7 +7,7 @@ > to preserve this first draft; the goal is to make it more accurate as the > project develops. -**Last reviewed:** 2026-08-03 +**Last reviewed:** 2026-08-04 **Current target:** Preserve the published `sota-v2` study as frozen historical evidence while pre-registering and rehearsing a finite `sota-v3` publication lane. @@ -23,13 +23,15 @@ artifact-integrity fixes landed in #85 as an explicit `sota-v3` contract. Sunday's merged work closed the contract-economics, same-view, gap-diagnostic, site-framing, statistical-tiering, and version-dispatched CI items. The current working tree contains a ten-model public-catalog cohort, a frozen 16-seed x -1-repeat statistical design, a provisional 4,096-token safety ceiling, an empty +1-repeat statistical design, a frozen 4,096-token smoke ceiling, an empty not-started smoke manifest, explicit runner dispatch, and a zero-spend synthetic -rehearsal. Exact authenticated routes, privacy acceptance, model-specific -reasoning compatibility, and private seed identity remain unresolved. All -route, spend, execution, and publication gates remain false. There is no real -v3 smoke or leaderboard artifact, and no paid v3 smoke or panel spend is -authorized. +rehearsal. Exact-route and synthetic-data privacy acceptance are recorded for +all ten routes, the private seed commitment is frozen with its secret in macOS +Keychain, and the latest authenticated route preflight plus Keychain-backed +dry-run pass without completion calls. Spend and strict-smoke execution are now +authorized under the committed $150 ceiling. Panel execution and publication +remain false until every strict smoke is accepted and the cap-pressure rule is +resolved. There is no real v3 smoke or leaderboard artifact yet. **Current weekly focus:** [#93 — v3 readiness program: consultant audit findings](https://github.com/nedcut/gm-bench/issues/93). Remaining [Issue #84](https://github.com/nedcut/gm-bench/issues/84) follow-through is @@ -632,12 +634,13 @@ snapshot is in [`docs/run_logs/sota-v3-design-amendment-2026-07-28.md`](run_logs/sota-v3-design-amendment-2026-07-28.md), superseding [`docs/run_logs/sota-v3-statistical-design-audit-2026-07-28.md`](run_logs/sota-v3-statistical-design-audit-2026-07-28.md). - Seed identity remains unfrozen: the private seed panel requires separate - owner authorization before generation with the audited unbiased generator, - and both execution gates test against the literal `frozen`, so provider - execution stays locked. The overall preregistration status remains - `provisional-pre-smoke` until route, privacy, reasoning, seed, and rehearsal - evidence is complete. Design amendment 2 (2026-08-03, + The owner-authorized private seed panel is now frozen: the audited unbiased + generator produced 16 high-entropy ordered seeds, only the salted hiding + commitment plus ordered execution hash are committed, and the secret values + and salt are escrowed in macOS Keychain. Exact-route and synthetic-data + privacy evidence are accepted, so the overall preregistration, registry, + protocol, pricing, output budget, spend, and strict-smoke gates are frozen. + Panel execution and publication remain locked. Design amendment 2 (2026-08-03, [`docs/run_logs/sota-v3-design-amendment-2026-08-03.md`](run_logs/sota-v3-design-amendment-2026-08-03.md)) is a pre-data cohort update: the OpenAI anchor moves from GPT-5.6 Luna Pro to plain GPT-5.6 Luna, and DeepSeek V4 Flash 0731 plus Tencent Hy3 join as @@ -651,9 +654,10 @@ snapshot is in lower bound 0.7941). The contract fingerprint is unchanged. The reproducible conservative pre-smoke reservation is recorded in `results/analysis/sota-v3-pre-smoke-cost-estimate.json`: 3,240 calls, - $89.845094 before contingency and $107.814113 at the committed 1.2x reserve. - Runtime remains pending accepted smoke telemetry, so this does not set an - operator ceiling or authorize spend. A + $106.073183 before contingency and $127.287820 at the committed 1.2x reserve. + Runtime remains pending accepted smoke telemetry. The owner-set $150 ceiling + and smoke-only spend authorization are now enforced; panel authorization is + still separate. A runtime private panel can change without changing the contract fingerprint; editing the canonical public leaderboard preset in `gm_bench/benchmark_config.py` does change it and would require one pre-data @@ -899,8 +903,8 @@ decision and why. | 2026-07-27 | Freeze v3 mechanics and reduce the pre-spend path to preregistration plus offline rehearsal. | PRs #92, #95, #98, #99, and #101 closed the mechanics, site-framing, same-view, version-dispatch, and claim-decomposition blockers. Continuing to add plausible realism changes now creates more schedule and evidence risk than it removes. The base SHA had no v3 lane/registry/manifest or v3 artifact; the working tree now has provisional fail-closed config files but still no selected model family or real/committed artifact. | Freeze score-affecting mechanics at `4f6ddddd6a6dd81c`; permit only one bounded, pre-data publication-parameter amendment if panel design requires it; preserve v2 literally; complete preregistration and a clean-checkout no-provider-call rehearsal; authorize no paid smoke or panel by this decision. | | 2026-07-30 | Reconcile the v3 pre-spend design across configs, policy, cost planning, seed commitment, and rehearsal. | The exact registered power procedure supports 15 independent seeds x 1 stochastic trajectory per model: base power 0.9461 and sensitivity power 0.8357 with Wilson lower bound 0.8283. One repeat is the registered estimand, not a dropped replicate. The prior configs disagreed on repeats and treated a live-readiness mismatch as non-fatal. No private seed was generated and no provider was called during reconciliation. | Bind the current lane to fingerprint `a523bdfcebe47bbd`, freeze the 15 x 1 statistical design, use a provisional 4,096/3,072/8,192 cap rule with whole-cohort invalidation and re-smoke on pressure, provide an unbiased private-seed generator, and make preregistration coherence a hard rehearsal gate. Keep every route, spend, execution, and publication authorization false. | | 2026-08-03 | Move the ten-model registry from `provisional-blocked` to `route-preflight-ready` while `evidence_state` is still pre-data. | Everything registered about the ten routes comes from the public OpenRouter catalog, which by the registry's own admission "does not prove authenticated exact-route access or provider privacy and retention behavior." The v2 lane already lost Nemotron 3 Ultra and DeepSeek V4 Pro to bounded HTTP 404s on routes that looked healthy publicly, so a failed authenticated probe is a live possibility, not a hypothetical. Route preflight is the cheapest test of that assumption: it makes zero completion calls and cannot launch a model subprocess, reserve spend, or create run state. Discovering a dead route now costs a JSON regeneration; discovering it after the seed panel is committed means a committed panel attached to a design that then changed, because cohort size drives the Holm family size, which drives the allocation and the reservation. | Registry `selection_status` becomes `route-preflight-ready`; `selection_frozen_at_utc` stays `null`. This is strictly weaker than `frozen` and unlocks nothing that costs money: measured against the live configs, route-preflight readiness goes from two blockers to one — the owner's separate `route_preflight_authorized` grant — while the smoke and panel phases stay at an identical 60 blockers, still including "provider execution is locked until the model registry is frozen." Asserted by `test_route_preflight_readiness_unlocks_nothing_that_costs_money`. Cohort identity is **not** frozen by this decision; freezing it remains a separate later decision informed by preflight results. Every lane authorization remains false. | - | 2026-08-03 | Grant `route_preflight_authorized`, run the authenticated zero-call route preflight, and correct the stale `qwen/qwen3.7-plus` endpoint tag. | The registered route metadata came from the public catalog, which cannot prove authenticated access. The probe makes zero completion calls, cannot launch a model subprocess, and cannot write run state — verified empirically, since the aborted first run left no files behind. It found exactly one defect in ten: the Alibaba endpoint tag for `qwen/qwen3.7-plus` had moved from `alibaba` to `alibaba/fp8`, while `provider_name`, `name`, status, and every published price stayed identical. | All ten routes now pass at $0.00 spend. `endpoint_tag` and `upstream_provider_slug` corrected in the registry, and the bound `provider_slug` in the pricing snapshot; the cost artifact regenerates byte-identically, so the reservation holds at $89.845094 / $107.814113. **The cohort stays at ten and the 16x1 allocation is unaffected** — a dead route would have forced a family-of-nine amendment and a power re-selection. `exact_route_acceptance` remains `unresolved`; smoke is still blocked by 60 issues, the same count as before the probe. `spend_authorized`, `smoke_execution_authorized`, `panel_execution_authorized`, and `publication_authorized` all remain false. Logged in [`docs/run_logs/sota-v3-route-preflight-2026-08-03.md`](run_logs/sota-v3-route-preflight-2026-08-03.md). | +| 2026-08-04 | Freeze the SOTA-v3 lane and authorize only the strict-smoke phase. | PR #110 is merged; the refreshed exact routes pass authenticated metadata checks. The private 16-seed panel was owner-authorized and generated before any v3 model result. OpenRouter's current policy distinguishes data-collection denial from ZDR: all routes run with `data_collection=deny`, while 5/10 exact routes are listed as ZDR. Grok and Mistral advertise `max_tokens` but omit a numeric `max_completion_tokens`, and no same-model alternative fixes that metadata gap. | Commit only the salted hiding commitment and ordered seed hash; escrow secret values in macOS Keychain. Accept retention for synthetic non-confidential benchmark inputs, prohibit provider training use, and record ZDR per route rather than claiming it universally. Permit the two null-cap routes only as `request-cap-pending-strict-smoke`; complete smoke telemetry remains mandatory before panel authorization. Set the lane, registry, protocol, and pricing to frozen; authorize spend and strict smokes under the $150 ceiling; leave panel and publication authorization false. Logged in [`docs/run_logs/sota-v3-smoke-readiness-freeze-2026-08-04.md`](run_logs/sota-v3-smoke-readiness-freeze-2026-08-04.md). | ## Experiment and release log diff --git a/docs/production_benchmark.md b/docs/production_benchmark.md index c99674a..0d4ecce 100644 --- a/docs/production_benchmark.md +++ b/docs/production_benchmark.md @@ -49,6 +49,25 @@ multi-megabyte failed artifact is intentionally not retained. Prefer `--preset smoke` first; a clean serial strict panel is multi-hour quota spend, not a quick retry. +For the frozen SOTA-v3 strict-smoke lane, do not export or paste the private +seed values. The owner-authorized panel is stored in macOS Keychain, and the +launcher verifies both its salted commitment and ordered execution hash before +passing it to the publication runner in-process. The operator must still type +an explicit ceiling: + +```bash +uv run python scripts/run_publication_matrix.py route-preflight \ + --contract sota-v3 +uv run python scripts/run_sota_v3_smoke_from_keychain.py \ + --max-spend-usd 150 +``` + +The route preflight makes zero completion calls and should be rerun immediately +before the paid smoke because health and pricing metadata drift. The launcher +always retains `GM_BENCH_WORKERS=1`, exact-route pinning, no fallbacks, strict +failure handling, and the committed reservation/ceiling checks. It authorizes +no panel work; panel execution remains a separate post-smoke decision. + Fresh-spawn serial model panels write an atomic checkpoint after every completed seed/repeat and stop after two consecutive adapter failures. Resume with `--resume` for the default checkpoint or add one or more `--resume-from PATH` diff --git a/docs/run_logs/sota-v3-smoke-readiness-freeze-2026-08-04.md b/docs/run_logs/sota-v3-smoke-readiness-freeze-2026-08-04.md new file mode 100644 index 0000000..e6da141 --- /dev/null +++ b/docs/run_logs/sota-v3-smoke-readiness-freeze-2026-08-04.md @@ -0,0 +1,88 @@ +# SOTA-v3 smoke-readiness freeze — 2026-08-04 + +## Outcome + +The SOTA-v3 lane is frozen and authorized for the **strict smoke phase only**. +No model completion was called while preparing this freeze. Panel execution and +publication remain locked. + +## Frozen inputs + +- Contract fingerprint: `a523bdfcebe47bbd`. +- Cohort: ten exact OpenRouter routes in `config/sota_v3_models.json`. +- Statistical allocation: 16 private seeds x 1 repeat. +- Output policy: common 4,096-token request cap; 3,072-token pressure trigger; + one symmetric 8,192-token fallback amendment allowed only if the whole smoke + family is invalidated and rerun. +- Strict failure handling, one protocol repair, no route fallbacks, JSON mode, + `data_collection=deny`, and serial `GM_BENCH_WORKERS=1` execution. +- Spend: smoke authorized under the committed `$150.00` operator ceiling; + panel spend is not authorized. + +## Private seed commitment + +The owner-authorized generator sampled 16 ordered, unique high-entropy 63-bit +seeds with `uniform-rejection-sampling-secrets-randbelow-63bit-v1`. + +- Ordered execution SHA-256: + `291fa61cc3dfd8b23fdd79cce3c80a0a98f918f6c8757d35c21b4d8131cc6099` +- Salted hiding commitment SHA-256: + `7f8da7ca4db4a698ea2b0506af8568c89744e18508d8dedbfeb1c87e90a2b5f8` +- Secret values and salt: stored only in macOS Keychain service + `gm-bench-sota-v3-private-panel`; the temporary plaintext file was securely + removed after a hash-only escrow verification. + +The repository contains no seed or salt values. The launcher +`scripts/run_sota_v3_smoke_from_keychain.py` verifies both committed hashes +before setting `GM_BENCH_PRIVATE_SEEDS` in-process. + +## Route and privacy evidence + +`results/analysis/sota-v3-route-acceptance-evidence.json` records an +authenticated credits-metadata success, all ten exact endpoint identities, +supported parameters, health telemetry, provider policy links, and the live +OpenRouter ZDR classification. The collector made zero completion calls and +retained no credential, balance, or account-usage value. + +The accepted privacy standard is explicit rather than implied: + +- GM-Bench sends synthetic game state with no personal or confidential data. +- `OPENROUTER_DATA_COLLECTION=deny` is required, so training-use routes are + excluded. +- Provider retention terms are accepted for this synthetic benchmark. +- ZDR is preferred but is not required; 5 of the 10 exact routes were present + in OpenRouter's authenticated ZDR endpoint list at the freeze. + +## Null cap metadata + +The exact Grok 4.5 xAI ZDR and Mistral Medium 3.5 Mistral routes advertise +`max_tokens` but return `max_completion_tokens: null`. No same-model alternative +route supplies a numeric maximum. They therefore carry the narrow status +`request-cap-pending-strict-smoke`. + +This is not a general missing-metadata bypass. The route gate accepts it only +when the endpoint advertises `max_tokens`, the registry declares the exact +exception, and strict-smoke verification remains required. Panel execution +still requires complete per-call finish-reason and usage telemetry, zero +truncations, and a peak below 3,072 tokens for every registered smoke. + +## Zero-spend verification + +- Authenticated route evidence collection: 10 routes, 5 ZDR, 0 completions. +- Route preflight: 10/10 passed, 0 completions. +- Keychain-backed smoke dry-run: all ten serial commands built successfully; + no child model process launched. +- The live Luna and GLM prices remained below the conservative undiscounted + snapshot; decreases were reported and allowed. + +Launch only with an explicit ceiling: + +```bash +uv run python scripts/run_sota_v3_smoke_from_keychain.py \ + --max-spend-usd 150 +``` + +Before launching, rerun the route preflight because health and pricing evidence +has an hours-long shelf life. After the smokes, record and validate every +artifact, evaluate the whole-cohort cap-pressure rule, refresh the cost/runtime +plan, and request a separate panel authorization. diff --git a/gm_bench/publication.py b/gm_bench/publication.py index 4df5844..13a063c 100644 --- a/gm_bench/publication.py +++ b/gm_bench/publication.py @@ -12,6 +12,7 @@ import json import math import re +from pathlib import Path from typing import Any PUBLICATION_FORMAT = "gm-bench-result-summary-v1" @@ -57,8 +58,19 @@ def raw_artifact_link_issues( "data_collection_policy_accepted", "retention_policy_accepted", "training_use_policy_accepted", - "zero_data_retention_policy_accepted", ) +PENDING_STRICT_SMOKE_CAP_VERIFICATION = { + "status": "request-cap-pending-strict-smoke", + "catalog_max_completion_tokens": None, + "request_parameter": "max_tokens", + "strict_smoke_required": True, +} + + +def is_pending_strict_smoke_cap(value: Any) -> bool: + """Return whether *value* is the exact fail-closed cap deferral contract.""" + + return isinstance(value, dict) and value == PENDING_STRICT_SMOKE_CAP_VERIFICATION def _registered_route_options(registry: dict[str, Any], model: dict[str, Any]) -> tuple[dict[str, str], list[str]]: @@ -92,6 +104,7 @@ def v3_route_identity_sha256(registry: dict[str, Any], model: dict[str, Any]) -> "output_token_cap": registry.get("output_token_cap"), "reasoning_policy": model.get("reasoning_policy"), "reasoning_effort": model.get("reasoning_effort"), + "output_cap_verification": model.get("output_cap_verification"), "supported_parameters": sorted(str(value) for value in model.get("catalog_supported_parameters") or []), "requested_options": requested, "absent_options": absent, @@ -114,8 +127,36 @@ def v3_route_acceptance_issues(registry: dict[str, Any]) -> list[str]: issues: list[str] = [] if acceptance.get("status") != "accepted": issues.append("sota-v3 exact-route acceptance status is not accepted") + privacy_standard = acceptance.get("privacy_standard") + if not isinstance(privacy_standard, dict): + issues.append("sota-v3 exact-route privacy standard is missing") + else: + if privacy_standard.get("data_classification") != "synthetic-benchmark-no-personal-or-confidential-data": + issues.append("sota-v3 exact-route privacy data classification is not accepted") + if privacy_standard.get("provider_data_collection") != "deny": + issues.append("sota-v3 exact-route privacy standard must deny provider data collection") + if privacy_standard.get("provider_training_use_allowed") is not False: + issues.append("sota-v3 exact-route privacy standard must prohibit provider training use") + if privacy_standard.get("zero_data_retention_required") is not False: + issues.append("sota-v3 exact-route privacy standard must explicitly resolve the ZDR requirement") entries = acceptance.get("entries") entries = entries if isinstance(entries, dict) else {} + evidence: dict[str, Any] | None = None + evidence_artifact = acceptance.get("evidence_artifact") + if not isinstance(evidence_artifact, str) or not evidence_artifact.strip(): + issues.append("sota-v3 exact-route evidence artifact is missing") + else: + try: + loaded = json.loads(Path(evidence_artifact).read_text()) + if not isinstance(loaded, dict): + raise ValueError("evidence artifact must contain a JSON object") + evidence = loaded + except (OSError, ValueError, json.JSONDecodeError) as exc: + issues.append(f"sota-v3 exact-route evidence artifact cannot be read: {exc}") + evidence_routes = evidence.get("routes") if evidence is not None else None + if evidence is not None and not isinstance(evidence_routes, dict): + issues.append("sota-v3 exact-route evidence artifact routes are missing") + evidence_routes = {} model_ids = {str(model.get("id") or "") for model in models} if set(entries) != model_ids: issues.append("sota-v3 exact-route acceptance entries must exactly match the registered model ids") @@ -136,6 +177,10 @@ def v3_route_acceptance_issues(registry: dict[str, Any]) -> list[str]: evidence_sha = entry.get("route_evidence_sha256") if not isinstance(evidence_sha, str) or re.fullmatch(r"[0-9a-f]{64}", evidence_sha) is None: issues.append(f"{prefix} authenticated route evidence digest is missing") + elif evidence is not None: + route = evidence_routes.get(model_id) if isinstance(evidence_routes, dict) else None + if not isinstance(route, dict) or evidence_sha != canonical_sha256(route): + issues.append(f"{prefix} authenticated route evidence digest does not match its canonical payload") privacy = entry.get("privacy_acceptance") if not isinstance(privacy, dict) or privacy.get("status") != "accepted": @@ -146,11 +191,30 @@ def v3_route_acceptance_issues(registry: dict[str, Any]) -> list[str]: for field in _PRIVACY_ACCEPTANCE_FIELDS: if privacy.get(field) is not True: issues.append(f"{prefix} {field} is not accepted") + zdr_endpoint = privacy.get("zero_data_retention_endpoint") + if not isinstance(zdr_endpoint, bool): + issues.append(f"{prefix} zero_data_retention_endpoint must be recorded as a boolean") + if privacy.get("zero_data_retention_requirement_satisfied") is not True: + issues.append(f"{prefix} zero_data_retention_requirement_satisfied is not accepted") if not isinstance(privacy.get("accepted_at_utc"), str) or not privacy["accepted_at_utc"].strip(): issues.append(f"{prefix} privacy acceptance timestamp is missing") privacy_sha = privacy.get("evidence_sha256") if not isinstance(privacy_sha, str) or re.fullmatch(r"[0-9a-f]{64}", privacy_sha) is None: issues.append(f"{prefix} privacy evidence digest is missing") + elif evidence is not None: + route = evidence_routes.get(model_id) if isinstance(evidence_routes, dict) else None + if not isinstance(route, dict): + issues.append(f"{prefix} privacy evidence digest does not match its canonical payload") + else: + privacy_evidence = { + "route_identity_sha256": route.get("route_identity_sha256"), + "privacy_standard": evidence.get("privacy_standard"), + "zero_data_retention_endpoint": route.get("zero_data_retention_endpoint"), + "provider_policy": route.get("provider_policy"), + "official_policy_sources": evidence.get("official_policy_sources"), + } + if privacy_sha != canonical_sha256(privacy_evidence): + issues.append(f"{prefix} privacy evidence digest does not match its canonical payload") return issues diff --git a/results/analysis/sota-v3-route-acceptance-evidence.json b/results/analysis/sota-v3-route-acceptance-evidence.json new file mode 100644 index 0000000..deaad12 --- /dev/null +++ b/results/analysis/sota-v3-route-acceptance-evidence.json @@ -0,0 +1,387 @@ +{ + "account_authentication": { + "method": "OpenRouter credits metadata endpoint returned success", + "sensitive_values_included": false, + "status": "authenticated" + }, + "completion_calls": 0, + "contract": "sota-v3", + "contract_fingerprint": "a523bdfcebe47bbd", + "format": "gm-bench-route-acceptance-evidence-v1", + "generated_at_utc": "2026-08-04T23:26:17+00:00", + "official_policy_sources": [ + "https://openrouter.ai/docs/guides/features/zdr", + "https://openrouter.ai/docs/guides/privacy/provider-logging/", + "https://openrouter.ai/docs/guides/routing/provider-selection", + "https://openrouter.ai/docs/api/api-reference/endpoints/list-endpoints-zdr" + ], + "privacy_standard": { + "data_classification": "synthetic-benchmark-no-personal-or-confidential-data", + "provider_data_collection": "deny", + "provider_retention_terms": "reviewed-and-accepted-for-synthetic-benchmark", + "provider_training_use_allowed": false, + "zero_data_retention_preferred": true, + "zero_data_retention_required": false + }, + "routes": { + "openrouter-claude-sonnet-5-bedrock": { + "authenticated_metadata_read": true, + "endpoint": { + "max_completion_tokens": 128000, + "name": "Amazon Bedrock | anthropic/claude-sonnet-5-20260630", + "provider_name": "Amazon Bedrock", + "status": 0, + "supported_parameters": [ + "include_reasoning", + "max_tokens", + "reasoning", + "reasoning_effort", + "response_format", + "stop", + "structured_outputs", + "tool_choice", + "tools", + "verbosity" + ], + "tag": "amazon-bedrock/global", + "uptime_last_1d": 99.84734119243491, + "uptime_last_30m": 100 + }, + "output_cap_verification_status": "endpoint-metadata-verified", + "provider_policy": { + "privacy_policy_url": "https://aws.amazon.com/privacy", + "provider_slug": "amazon-bedrock", + "terms_of_service_url": "https://aws.amazon.com/service-terms/" + }, + "route_identity_sha256": "cd14b14e915dbac2656c1483972fb585df7d718635f2748e044ecfe1d0c6be4c", + "zero_data_retention_endpoint": true + }, + "openrouter-deepseek-v4-flash-0731-cloudflare": { + "authenticated_metadata_read": true, + "endpoint": { + "max_completion_tokens": 384000, + "name": "Cloudflare | deepseek/deepseek-v4-flash-20260731", + "provider_name": "Cloudflare", + "status": 0, + "supported_parameters": [ + "frequency_penalty", + "include_reasoning", + "logit_bias", + "logprobs", + "max_tokens", + "min_p", + "presence_penalty", + "reasoning", + "reasoning_effort", + "repetition_penalty", + "response_format", + "seed", + "stop", + "structured_outputs", + "temperature", + "tool_choice", + "tools", + "top_k", + "top_logprobs", + "top_p" + ], + "tag": "cloudflare/fp8", + "uptime_last_1d": 99.20383493778141, + "uptime_last_30m": 99.57471943295924 + }, + "output_cap_verification_status": "endpoint-metadata-verified", + "provider_policy": { + "privacy_policy_url": "https://developers.cloudflare.com/workers-ai/privacy", + "provider_slug": "cloudflare", + "terms_of_service_url": "https://www.cloudflare.com/service-specific-terms-developer-platform/#developer-platform-terms" + }, + "route_identity_sha256": "ee801d3d09aa511f61eb5acde13825f633b33da5eaee3a3c3090f51f7914eac5", + "zero_data_retention_endpoint": false + }, + "openrouter-gemini-3.6-flash-google-ai-studio": { + "authenticated_metadata_read": true, + "endpoint": { + "max_completion_tokens": 65536, + "name": "Google AI Studio | google/gemini-3.6-flash-20260721", + "provider_name": "Google AI Studio", + "status": 0, + "supported_parameters": [ + "include_reasoning", + "max_tokens", + "reasoning", + "reasoning_effort", + "response_format", + "seed", + "structured_outputs", + "temperature", + "tool_choice", + "tools", + "top_p" + ], + "tag": "google-ai-studio", + "uptime_last_1d": 98.47300897834896, + "uptime_last_30m": 98.37067209775967 + }, + "output_cap_verification_status": "endpoint-metadata-verified", + "provider_policy": { + "privacy_policy_url": "https://cloud.google.com/terms/cloud-privacy-notice", + "provider_slug": "google-ai-studio", + "terms_of_service_url": "https://cloud.google.com/terms/" + }, + "route_identity_sha256": "2cab5c7c0f89d2de1c2b2dae5dbe277d1e1a20f4649c5b42e717b9545f830558", + "zero_data_retention_endpoint": false + }, + "openrouter-glm-5.2-novita": { + "authenticated_metadata_read": true, + "endpoint": { + "max_completion_tokens": 131072, + "name": "Novita | z-ai/glm-5.2-20260616", + "provider_name": "Novita", + "status": 0, + "supported_parameters": [ + "frequency_penalty", + "include_reasoning", + "max_tokens", + "presence_penalty", + "reasoning", + "reasoning_effort", + "repetition_penalty", + "response_format", + "seed", + "stop", + "temperature", + "tool_choice", + "tools", + "top_k", + "top_p" + ], + "tag": "novita/fp8", + "uptime_last_1d": 99.12393326496591, + "uptime_last_30m": 97.84186221845255 + }, + "output_cap_verification_status": "endpoint-metadata-verified", + "provider_policy": { + "privacy_policy_url": "https://novita.ai/legal/privacy-policy", + "provider_slug": "novita", + "terms_of_service_url": "https://novita.ai/legal/terms-of-service" + }, + "route_identity_sha256": "864af6547ed9669d9f1e16c9cf93ce86c3c341f3383a21750fe0228b63c4d20d", + "zero_data_retention_endpoint": true + }, + "openrouter-gpt-5.6-luna-openai": { + "authenticated_metadata_read": true, + "endpoint": { + "max_completion_tokens": 128000, + "name": "OpenAI | openai/gpt-5.6-luna-20260709", + "provider_name": "OpenAI", + "status": 0, + "supported_parameters": [ + "include_reasoning", + "max_tokens", + "reasoning", + "reasoning_effort", + "response_format", + "seed", + "structured_outputs", + "tool_choice", + "tools" + ], + "tag": "openai", + "uptime_last_1d": 99.25235247330117, + "uptime_last_30m": 99.02350802859962 + }, + "output_cap_verification_status": "endpoint-metadata-verified", + "provider_policy": { + "privacy_policy_url": "https://openai.com/policies/privacy-policy/", + "provider_slug": "openai", + "terms_of_service_url": "https://openai.com/policies/row-terms-of-use/" + }, + "route_identity_sha256": "7968d261fc1919563259c9226842ac936f1b0a9b1250f4b4b3013c3b6ebbdabb", + "zero_data_retention_endpoint": false + }, + "openrouter-grok-4.5-xai": { + "authenticated_metadata_read": true, + "endpoint": { + "max_completion_tokens": null, + "name": "xAI | x-ai/grok-4.5-20260708", + "provider_name": "xAI", + "status": 0, + "supported_parameters": [ + "frequency_penalty", + "include_reasoning", + "logprobs", + "max_tokens", + "presence_penalty", + "reasoning", + "reasoning_effort", + "response_format", + "seed", + "stop", + "structured_outputs", + "temperature", + "tool_choice", + "tools", + "top_logprobs", + "top_p" + ], + "tag": "xai/zdr", + "uptime_last_1d": 99.95176825658132, + "uptime_last_30m": 100 + }, + "output_cap_verification_status": "request-cap-pending-strict-smoke", + "provider_policy": { + "privacy_policy_url": "https://x.ai/legal/privacy-policy", + "provider_slug": "xai", + "terms_of_service_url": "https://x.ai/legal/terms-of-service-enterprise" + }, + "route_identity_sha256": "750efc6eba3496bc94f7138cd410c5a085f570e2d476cdeecf434f9340d761b1", + "zero_data_retention_endpoint": true + }, + "openrouter-hy3-tencent": { + "authenticated_metadata_read": true, + "endpoint": { + "max_completion_tokens": 128000, + "name": "Tencent | tencent/hy3-20260706", + "provider_name": "Tencent", + "status": 0, + "supported_parameters": [ + "include_reasoning", + "max_completion_tokens", + "max_tokens", + "reasoning", + "reasoning_effort", + "response_format", + "stop", + "temperature", + "tool_choice", + "tools" + ], + "tag": "tencent/fp8", + "uptime_last_1d": 99.88669313467457, + "uptime_last_30m": 99.8017839444995 + }, + "output_cap_verification_status": "endpoint-metadata-verified", + "provider_policy": { + "privacy_policy_url": "https://www.tencentcloud.com/en/document/product/1300/78952", + "provider_slug": "tencent", + "terms_of_service_url": "https://www.tencentcloud.com/en/document/product/301/78869" + }, + "route_identity_sha256": "845a8c55b92983428fe7e3d75aecaf07172f2c84c57b43ecbbb0405a51ad8ee4", + "zero_data_retention_endpoint": true + }, + "openrouter-minimax-m3-deepinfra": { + "authenticated_metadata_read": true, + "endpoint": { + "max_completion_tokens": 512000, + "name": "DeepInfra | minimax/minimax-m3-20260531", + "provider_name": "DeepInfra", + "status": 0, + "supported_parameters": [ + "frequency_penalty", + "include_reasoning", + "logit_bias", + "max_tokens", + "min_p", + "presence_penalty", + "reasoning", + "repetition_penalty", + "response_format", + "seed", + "stop", + "temperature", + "tool_choice", + "tools", + "top_k", + "top_p" + ], + "tag": "deepinfra/fp8", + "uptime_last_1d": 99.5721639656816, + "uptime_last_30m": 100 + }, + "output_cap_verification_status": "endpoint-metadata-verified", + "provider_policy": { + "privacy_policy_url": "https://deepinfra.com/privacy", + "provider_slug": "deepinfra", + "terms_of_service_url": "https://deepinfra.com/terms" + }, + "route_identity_sha256": "3a5ff981f3b2ce8e13604c9a2c5ee3a3794b5dbd09228a11c641ed33bf2ec326", + "zero_data_retention_endpoint": true + }, + "openrouter-mistral-medium-3.5-mistral": { + "authenticated_metadata_read": true, + "endpoint": { + "max_completion_tokens": null, + "name": "Mistral | mistralai/mistral-medium-3.5-20260430", + "provider_name": "Mistral", + "status": 0, + "supported_parameters": [ + "frequency_penalty", + "include_reasoning", + "max_tokens", + "presence_penalty", + "reasoning", + "reasoning_effort", + "response_format", + "seed", + "stop", + "structured_outputs", + "temperature", + "tool_choice", + "tools", + "top_p" + ], + "tag": "mistral", + "uptime_last_1d": 99.97488665952919, + "uptime_last_30m": 99.87012987012987 + }, + "output_cap_verification_status": "request-cap-pending-strict-smoke", + "provider_policy": { + "privacy_policy_url": "https://mistral.ai/terms/#privacy-policy", + "provider_slug": "mistral", + "terms_of_service_url": "https://mistral.ai/terms/#terms-of-use" + }, + "route_identity_sha256": "6b50860e44b7c8c4e52ac53faa23cebffec2f4f430fac1cd0aaf1a80683b3a03", + "zero_data_retention_endpoint": false + }, + "openrouter-qwen3.8-max-alibaba": { + "authenticated_metadata_read": true, + "endpoint": { + "max_completion_tokens": 131072, + "name": "Alibaba | qwen/qwen3.8-max-20260803", + "provider_name": "Alibaba", + "status": 0, + "supported_parameters": [ + "frequency_penalty", + "include_reasoning", + "logprobs", + "max_tokens", + "presence_penalty", + "reasoning", + "reasoning_effort", + "response_format", + "seed", + "stop", + "structured_outputs", + "temperature", + "tool_choice", + "tools", + "top_k", + "top_logprobs", + "top_p" + ], + "tag": "alibaba", + "uptime_last_1d": 99.9995100801019, + "uptime_last_30m": 100 + }, + "output_cap_verification_status": "endpoint-metadata-verified", + "provider_policy": { + "privacy_policy_url": "https://www.alibabacloud.com/help/en/legal/latest/alibaba-cloud-international-website-privacy-policy", + "provider_slug": "alibaba", + "terms_of_service_url": "https://www.alibabacloud.com/help/en/legal/latest/alibaba-cloud-international-website-product-terms-of-service-v-3-8-0" + }, + "route_identity_sha256": "ad5446583e6d608032760931054c42a8b774149e3c3cf6dd6426656b18871804", + "zero_data_retention_endpoint": false + } + }, + "schema_version": 1 +} diff --git a/scripts/collect_sota_v3_route_evidence.py b/scripts/collect_sota_v3_route_evidence.py new file mode 100644 index 0000000..2801075 --- /dev/null +++ b/scripts/collect_sota_v3_route_evidence.py @@ -0,0 +1,269 @@ +#!/usr/bin/env python3 +"""Collect authenticated, zero-completion route and privacy evidence for SOTA-v3. + +The collector reads only OpenRouter metadata endpoints. It never sends a model +prompt, never creates a completion, and never records the account balance or API +key. With ``--apply-registry`` it freezes exact-route acceptance against the +generated evidence artifact after every registered route passes. +""" + +from __future__ import annotations + +import argparse +import http.client +import json +import os +import sys +import urllib.parse +from datetime import datetime, timezone +from pathlib import Path +from typing import Any + +ROOT = Path(__file__).resolve().parents[1] +if str(ROOT) not in sys.path: + sys.path.insert(0, str(ROOT)) + +from gm_bench.environment import load_environment_files # noqa: E402 +from gm_bench.publication import ( # noqa: E402 + canonical_sha256, + is_pending_strict_smoke_cap, + v3_route_identity_sha256, +) + +PRIVACY_STANDARD = { + "data_classification": "synthetic-benchmark-no-personal-or-confidential-data", + "provider_data_collection": "deny", + "provider_training_use_allowed": False, + "provider_retention_terms": "reviewed-and-accepted-for-synthetic-benchmark", + "zero_data_retention_required": False, + "zero_data_retention_preferred": True, +} +POLICY_SOURCES = [ + "https://openrouter.ai/docs/guides/features/zdr", + "https://openrouter.ai/docs/guides/privacy/provider-logging/", + "https://openrouter.ai/docs/guides/routing/provider-selection", + "https://openrouter.ai/docs/api/api-reference/endpoints/list-endpoints-zdr", +] + + +def _read_json(path: Path) -> dict[str, Any]: + payload = json.loads(path.read_text()) + if not isinstance(payload, dict): + raise ValueError(f"{path} must contain a JSON object") + return payload + + +def _get_json(path: str, headers: dict[str, str]) -> dict[str, Any]: + parsed = urllib.parse.urlsplit(path) + if parsed.scheme or parsed.netloc or not parsed.path.startswith("/api/v1/") or parsed.path != path: + raise ValueError(f"refusing non-OpenRouter metadata path: {path!r}") + # The TLS host is a literal and the request path is constrained to /api/v1/ above. + connection = http.client.HTTPSConnection( # nosemgrep: python.lang.security.audit.httpsconnection-detected.httpsconnection-detected + "openrouter.ai", timeout=30 + ) + try: + connection.request("GET", path, headers=headers) + response = connection.getresponse() + if response.status < 200 or response.status >= 300: + raise RuntimeError(f"OpenRouter metadata request failed with HTTP {response.status}") + payload = json.load(response) + finally: + connection.close() + if not isinstance(payload, dict): + raise ValueError(f"{path} did not return a JSON object") + return payload + + +def _exact_endpoint(model: dict[str, Any], endpoints: list[dict[str, Any]]) -> dict[str, Any]: + matches = [ + endpoint + for endpoint in endpoints + if endpoint.get("provider_name") == model.get("upstream_provider") + and endpoint.get("tag") == model.get("endpoint_tag") + and endpoint.get("name") == model.get("endpoint_name") + and endpoint.get("status") == 0 + ] + if len(matches) != 1: + raise ValueError(f"{model.get('id')}: expected one healthy exact endpoint, found {len(matches)}") + return matches[0] + + +def _public_endpoint_record(endpoint: dict[str, Any]) -> dict[str, Any]: + return { + "provider_name": endpoint.get("provider_name"), + "tag": endpoint.get("tag"), + "name": endpoint.get("name"), + "status": endpoint.get("status"), + "max_completion_tokens": endpoint.get("max_completion_tokens"), + "supported_parameters": sorted(str(value) for value in endpoint.get("supported_parameters") or []), + "uptime_last_30m": endpoint.get("uptime_last_30m"), + "uptime_last_1d": endpoint.get("uptime_last_1d"), + } + + +def collect(registry: dict[str, Any], headers: dict[str, str]) -> dict[str, Any]: + # A successful authenticated credits read proves the bearer token is valid; + # no balance or usage value is retained in the evidence artifact. + _get_json("/api/v1/credits", headers) + zdr_rows = _get_json("/api/v1/endpoints/zdr", headers).get("data") or [] + providers = _get_json("/api/v1/providers", headers).get("data") or [] + provider_by_slug = {str(row.get("slug")): row for row in providers if isinstance(row, dict)} + zdr_identities = { + (row.get("model_id"), row.get("provider_name"), row.get("tag"), row.get("name")) + for row in zdr_rows + if isinstance(row, dict) + } + generated_at = datetime.now(timezone.utc).isoformat(timespec="seconds") + routes: dict[str, Any] = {} + cap = registry.get("output_token_cap") + for model in registry.get("models") or []: + model_id = str(model["id"]) + model_path = urllib.parse.quote(str(model["model"]), safe="/") + endpoint_payload = _get_json(f"/api/v1/models/{model_path}/endpoints", headers) + endpoint = _exact_endpoint(model, list((endpoint_payload.get("data") or {}).get("endpoints") or [])) + supported = set(endpoint.get("supported_parameters") or []) + required = {"max_tokens", "reasoning", "response_format"} + if not required <= supported: + raise ValueError(f"{model_id}: exact endpoint is missing {sorted(required - supported)}") + maximum = endpoint.get("max_completion_tokens") + if isinstance(maximum, int) and not isinstance(maximum, bool) and isinstance(cap, int) and maximum >= cap: + cap_status = "endpoint-metadata-verified" + elif maximum is None and is_pending_strict_smoke_cap(model.get("output_cap_verification")): + cap_status = "request-cap-pending-strict-smoke" + else: + raise ValueError(f"{model_id}: exact endpoint does not establish the registered {cap}-token cap") + slug = str(model["upstream_provider_slug"]).split("/", 1)[0] + provider = provider_by_slug.get(slug) + if not isinstance(provider, dict): + raise ValueError(f"{model_id}: OpenRouter provider policy record {slug!r} is missing") + identity = ( + model.get("model"), + model.get("upstream_provider"), + model.get("endpoint_tag"), + model.get("endpoint_name"), + ) + route_identity = v3_route_identity_sha256(registry, model) + routes[model_id] = { + "route_identity_sha256": route_identity, + "endpoint": _public_endpoint_record(endpoint), + "authenticated_metadata_read": True, + "output_cap_verification_status": cap_status, + "zero_data_retention_endpoint": identity in zdr_identities, + "provider_policy": { + "provider_slug": slug, + "privacy_policy_url": provider.get("privacy_policy_url"), + "terms_of_service_url": provider.get("terms_of_service_url"), + }, + } + return { + "format": "gm-bench-route-acceptance-evidence-v1", + "schema_version": 1, + "contract": registry.get("contract"), + "contract_fingerprint": registry.get("contract_fingerprint"), + "generated_at_utc": generated_at, + "completion_calls": 0, + "account_authentication": { + "status": "authenticated", + "method": "OpenRouter credits metadata endpoint returned success", + "sensitive_values_included": False, + }, + "official_policy_sources": POLICY_SOURCES, + "privacy_standard": PRIVACY_STANDARD, + "routes": routes, + } + + +def apply_registry(registry: dict[str, Any], evidence: dict[str, Any], evidence_path: Path) -> dict[str, Any]: + accepted_at = str(evidence["generated_at_utc"]) + entries: dict[str, Any] = {} + for model in registry.get("models") or []: + model_id = str(model["id"]) + route = evidence["routes"][model_id] + privacy_evidence = { + "route_identity_sha256": route["route_identity_sha256"], + "privacy_standard": evidence["privacy_standard"], + "zero_data_retention_endpoint": route["zero_data_retention_endpoint"], + "provider_policy": route["provider_policy"], + "official_policy_sources": evidence["official_policy_sources"], + } + entries[model_id] = { + "route_identity_sha256": route["route_identity_sha256"], + "authenticated": True, + "verified_at_utc": accepted_at, + "route_evidence_sha256": canonical_sha256(route), + "privacy_acceptance": { + "status": "accepted", + "route_identity_sha256": route["route_identity_sha256"], + "data_collection_policy_accepted": True, + "retention_policy_accepted": True, + "training_use_policy_accepted": True, + "zero_data_retention_endpoint": route["zero_data_retention_endpoint"], + "zero_data_retention_requirement_satisfied": True, + "accepted_at_utc": accepted_at, + "evidence_sha256": canonical_sha256(privacy_evidence), + }, + } + registry["selection_status"] = "frozen" + registry["selection_frozen_at_utc"] = accepted_at + registry["exact_route_acceptance"] = { + "schema_version": 2, + "status": "accepted", + "accepted_at_utc": accepted_at, + "evidence_artifact": str(evidence_path.relative_to(ROOT)), + "privacy_standard": evidence["privacy_standard"], + "entries": entries, + } + return registry + + +def _write_json(path: Path, payload: dict[str, Any], *, sort_keys: bool = True) -> None: + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(json.dumps(payload, indent=2, sort_keys=sort_keys, allow_nan=False) + "\n") + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--registry", type=Path, default=ROOT / "config" / "sota_v3_models.json") + parser.add_argument( + "--output", + type=Path, + default=ROOT / "results" / "analysis" / "sota-v3-route-acceptance-evidence.json", + ) + parser.add_argument("--apply-registry", action="store_true") + args = parser.parse_args(argv) + root = ROOT.resolve() + output = args.output.resolve() + registry_path = args.registry.resolve() + for label, path in (("evidence output", output), ("registry", registry_path)): + if not path.is_relative_to(root): + parser.error(f"{label} must be inside the repository") + load_environment_files(ROOT) + key = os.environ.get("OPENROUTER_API_KEY") + if not key: + parser.error("OPENROUTER_API_KEY is required") + headers = { + "Authorization": f"Bearer {key}", + "User-Agent": "gm-bench-route-evidence/1", + } + registry = _read_json(registry_path) + evidence = collect(registry, headers) + _write_json(output, evidence) + if args.apply_registry: + _write_json(registry_path, apply_registry(registry, evidence, output), sort_keys=False) + print( + json.dumps( + { + "status": "accepted" if args.apply_registry else "collected", + "completion_calls": 0, + "routes": len(evidence["routes"]), + "zdr_routes": sum(route["zero_data_retention_endpoint"] for route in evidence["routes"].values()), + "output": str(output.relative_to(ROOT)), + }, + sort_keys=True, + ) + ) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/run_publication_matrix.py b/scripts/run_publication_matrix.py index f8f09a6..d6ced4b 100644 --- a/scripts/run_publication_matrix.py +++ b/scripts/run_publication_matrix.py @@ -78,6 +78,7 @@ from gm_bench.protocol import PHASES # noqa: E402 from gm_bench.publication import ( # noqa: E402 SMOKE_MANIFEST_FORMAT, + is_pending_strict_smoke_cap, publication_execution_issues, smoke_manifest_issues, ) @@ -96,6 +97,7 @@ class Cell: upstream_provider: str endpoint_tag: str endpoint_name: str + output_cap_verification: dict[str, Any] fixed_options: dict[str, str] absent_options: tuple[str, ...] @@ -230,6 +232,17 @@ def _validate_models( raise ValueError( f"publication registry is {expected_provider}-only but contains provider(s): {', '.join(mismatched)}" ) + for model in models: + cap_verification = model.get("output_cap_verification") + if cap_verification is not None: + if not is_pending_strict_smoke_cap(cap_verification): + raise ValueError( + f"publication model {model.get('id')!r} has an invalid output-cap verification exception" + ) + if "max_tokens" not in set(model.get("catalog_supported_parameters") or []): + raise ValueError( + f"publication model {model.get('id')!r} cannot defer cap verification without max_tokens support" + ) if not exact_routes: return for model in models: @@ -317,6 +330,7 @@ def build_cells(phase: str, model_id: str | None = None, cap: int | None = None) upstream_provider=str(model["upstream_provider"]), endpoint_tag=str(model["endpoint_tag"]), endpoint_name=str(model.get("endpoint_name") or ""), + output_cap_verification=dict(model.get("output_cap_verification") or {}), fixed_options=_registered_fixed_options(config, model), absent_options=_registered_absent_options(config, model), ) @@ -345,6 +359,7 @@ def build_cells(phase: str, model_id: str | None = None, cap: int | None = None) upstream_provider=str(model["upstream_provider"]), endpoint_tag=str(model.get("endpoint_tag") or ""), endpoint_name=str(model.get("endpoint_name") or ""), + output_cap_verification=dict(model.get("output_cap_verification") or {}), fixed_options={str(key): str(value) for key, value in (model.get("fixed_options") or {}).items()}, absent_options=tuple(str(value) for value in model.get("absent_options") or []), ) @@ -392,6 +407,7 @@ def build_cells(phase: str, model_id: str | None = None, cap: int | None = None) upstream_provider=str(model["upstream_provider"]), endpoint_tag=str(model["endpoint_tag"]), endpoint_name=str(model.get("endpoint_name") or ""), + output_cap_verification=dict(model.get("output_cap_verification") or {}), fixed_options=_registered_fixed_options(config, model), absent_options=_registered_absent_options(config, model), ) @@ -532,6 +548,8 @@ def _endpoint_issues(cell: Cell, payload: dict[str, Any]) -> list[str]: cap_fits = cell.cap is None or ( isinstance(maximum, int) and not isinstance(maximum, bool) and cell.cap <= maximum ) + if cell.cap is not None and maximum is None: + cap_fits = is_pending_strict_smoke_cap(cell.output_cap_verification) and "max_tokens" in supported if required <= supported and cap_fits: capable.append(endpoint) if not capable: diff --git a/scripts/run_sota_v3_smoke_from_keychain.py b/scripts/run_sota_v3_smoke_from_keychain.py new file mode 100644 index 0000000..516d233 --- /dev/null +++ b/scripts/run_sota_v3_smoke_from_keychain.py @@ -0,0 +1,107 @@ +#!/usr/bin/env python3 +"""Launch the authorized SOTA-v3 smoke without exposing private seeds. + +The private panel is read from macOS Keychain, verified against the committed +ordered hash and salted hiding commitment, and passed to the publication runner +only through ``GM_BENCH_PRIVATE_SEEDS`` in this process. The runner still +requires an explicit spend ceiling and retains every normal route, reservation, +strict-failure, and smoke-manifest gate. +""" + +from __future__ import annotations + +import argparse +import json +import os +import subprocess +import sys +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[1] +if str(ROOT) not in sys.path: + sys.path.insert(0, str(ROOT)) + +from gm_bench.benchmark_config import PRIVATE_SEEDS_ENV, seed_panel_hash # noqa: E402 +from scripts.run_publication_matrix import main as publication_main # noqa: E402 +from scripts.seed_panel_commitment import commitment, parse_ordered_seeds # noqa: E402 + +KEYCHAIN_ACCOUNT = "nedcutler" +KEYCHAIN_SERVICE = "gm-bench-sota-v3-private-panel" + + +def _keychain_record() -> dict[str, object]: + result = subprocess.run( # noqa: S603 - fixed macOS Keychain command + [ + "security", + "find-generic-password", + "-s", + KEYCHAIN_SERVICE, + "-a", + KEYCHAIN_ACCOUNT, + "-w", + ], + check=True, + capture_output=True, + text=True, + ) + raw = result.stdout.strip() + try: + decoded = bytes.fromhex(raw).decode() if not raw.startswith("{") else raw + record = json.loads(decoded) + except (UnicodeDecodeError, ValueError, json.JSONDecodeError) as exc: + raise ValueError("Keychain seed record is not valid gm-bench JSON") from exc + if not isinstance(record, dict): + raise ValueError("Keychain seed record must be a JSON object") + return record + + +def _verified_seed_text() -> str: + lane = json.loads((ROOT / "config" / "sota_v3_lane.json").read_text()) + panel = lane.get("seed_panel") or {} + record = _keychain_record() + seeds_text = record.get("seeds") + salt = record.get("salt") + if not isinstance(seeds_text, str) or not isinstance(salt, str): + raise ValueError("Keychain seed record is missing seeds or salt") + seeds = parse_ordered_seeds(seeds_text) + if panel.get("status") != "frozen" or panel.get("name") != "private-env": + raise ValueError("committed SOTA-v3 lane does not declare a frozen private panel") + if len(seeds) != panel.get("count") or seed_panel_hash(seeds) != panel.get("sha256"): + raise ValueError("Keychain seed order does not match the committed execution hash") + if commitment(salt, seeds) != panel.get("hiding_commitment_sha256"): + raise ValueError("Keychain seed panel does not match the committed hiding commitment") + return seeds_text + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--max-spend-usd", required=True, type=float) + parser.add_argument("--run-dir", default=str(ROOT / "data" / "publication" / "sota-v3-smokes")) + parser.add_argument("--model-id") + parser.add_argument("--dry-run", action="store_true") + parser.add_argument("--preflight-only", action="store_true") + args = parser.parse_args(argv) + os.environ[PRIVATE_SEEDS_ENV] = _verified_seed_text() + runner_args = [ + "smoke", + "--contract", + "sota-v3", + "--run-dir", + args.run_dir, + "--max-spend-usd", + str(args.max_spend_usd), + ] + if args.model_id: + runner_args.extend(["--model-id", args.model_id]) + if args.dry_run: + runner_args.append("--dry-run") + if args.preflight_only: + runner_args.append("--preflight-only") + try: + return publication_main(runner_args) + finally: + os.environ.pop(PRIVATE_SEEDS_ENV, None) + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tests/test_publication_release.py b/tests/test_publication_release.py index e3ac523..083ebcb 100644 --- a/tests/test_publication_release.py +++ b/tests/test_publication_release.py @@ -100,6 +100,12 @@ def _v3_release_fixture(tmp_path: Path) -> tuple[Path, Path]: route_identity = v3_route_identity_sha256(registry, model) registry["exact_route_acceptance"] = { "status": "accepted", + "privacy_standard": { + "data_classification": "synthetic-benchmark-no-personal-or-confidential-data", + "provider_data_collection": "deny", + "provider_training_use_allowed": False, + "zero_data_retention_required": False, + }, "entries": { "demo-v3": { "route_identity_sha256": route_identity, @@ -112,13 +118,38 @@ def _v3_release_fixture(tmp_path: Path) -> tuple[Path, Path]: "data_collection_policy_accepted": True, "retention_policy_accepted": True, "training_use_policy_accepted": True, - "zero_data_retention_policy_accepted": True, + "zero_data_retention_endpoint": True, + "zero_data_retention_requirement_satisfied": True, "accepted_at_utc": "2026-07-30T00:00:00Z", "evidence_sha256": "e" * 64, }, } }, } + route = { + "route_identity_sha256": route_identity, + "zero_data_retention_endpoint": True, + "provider_policy": {}, + } + evidence = { + "privacy_standard": registry["exact_route_acceptance"]["privacy_standard"], + "official_policy_sources": [], + "routes": {"demo-v3": route}, + } + entry = registry["exact_route_acceptance"]["entries"]["demo-v3"] + entry["route_evidence_sha256"] = canonical_sha256(route) + entry["privacy_acceptance"]["evidence_sha256"] = canonical_sha256( + { + "route_identity_sha256": route_identity, + "privacy_standard": evidence["privacy_standard"], + "zero_data_retention_endpoint": True, + "provider_policy": {}, + "official_policy_sources": [], + } + ) + evidence_path = repo / "results/analysis/route-evidence.json" + _write_json(evidence_path, evidence) + registry["exact_route_acceptance"]["evidence_artifact"] = str(evidence_path) seed_count = 6 lane = { "contract": "sota-v3", diff --git a/tests/test_publication_runner.py b/tests/test_publication_runner.py index e878c1b..908b07d 100644 --- a/tests/test_publication_runner.py +++ b/tests/test_publication_runner.py @@ -14,7 +14,12 @@ import scripts.run_publication_matrix as publication_runner from gm_bench.contract import BENCHMARK_VERSION, contract_fingerprint, scaffold_fingerprint -from gm_bench.publication import v3_route_acceptance_issues, v3_route_identity_sha256 +from gm_bench.publication import ( + PENDING_STRICT_SMOKE_CAP_VERIFICATION, + canonical_sha256, + v3_route_acceptance_issues, + v3_route_identity_sha256, +) from scripts.run_publication_matrix import ( _artifact_spend_usd, _cell_reservation_usd, @@ -69,6 +74,12 @@ def _frozen_panel_files( registry["exact_route_acceptance"] = { "schema_version": 1, "status": "accepted", + "privacy_standard": { + "data_classification": "synthetic-benchmark-no-personal-or-confidential-data", + "provider_data_collection": "deny", + "provider_training_use_allowed": False, + "zero_data_retention_required": False, + }, "entries": { model["id"]: { "route_identity_sha256": v3_route_identity_sha256(registry, model), @@ -81,7 +92,8 @@ def _frozen_panel_files( "data_collection_policy_accepted": True, "retention_policy_accepted": True, "training_use_policy_accepted": True, - "zero_data_retention_policy_accepted": True, + "zero_data_retention_endpoint": True, + "zero_data_retention_requirement_satisfied": True, "accepted_at_utc": "2026-07-30T00:00:00Z", "evidence_sha256": "e" * 64, }, @@ -89,6 +101,31 @@ def _frozen_panel_files( for model in registry["models"] }, } + evidence = { + "privacy_standard": registry["exact_route_acceptance"]["privacy_standard"], + "official_policy_sources": [], + "routes": {}, + } + for model_id, entry in registry["exact_route_acceptance"]["entries"].items(): + route = { + "route_identity_sha256": entry["route_identity_sha256"], + "zero_data_retention_endpoint": True, + "provider_policy": {}, + } + evidence["routes"][model_id] = route + entry["route_evidence_sha256"] = canonical_sha256(route) + entry["privacy_acceptance"]["evidence_sha256"] = canonical_sha256( + { + "route_identity_sha256": route["route_identity_sha256"], + "privacy_standard": evidence["privacy_standard"], + "zero_data_retention_endpoint": True, + "provider_policy": {}, + "official_policy_sources": [], + } + ) + evidence_path = tmp_path / "route-evidence.json" + evidence_path.write_text(json.dumps(evidence)) + registry["exact_route_acceptance"]["evidence_artifact"] = str(evidence_path) private_seeds = list(range(101, 110)) monkeypatch.setenv("GM_BENCH_PRIVATE_SEEDS", ",".join(str(seed) for seed in private_seeds)) lane = json.loads(Path("config/sota_v2_lane.json").read_text()) @@ -511,6 +548,15 @@ def test_validate_models_rejects_mandatory_minimum_policy_with_no_effort_declare publication_runner._validate_models([model]) +def test_validate_sweep_models_rejects_invalid_cap_deferral() -> None: + model = _minimal_registered_model( + output_cap_verification=dict(PENDING_STRICT_SMOKE_CAP_VERIFICATION), + catalog_supported_parameters=[], + ) + with pytest.raises(ValueError, match="cannot defer cap verification"): + publication_runner._validate_models([model], exact_routes=False) + + def test_committed_v2_panel_is_locked_after_current_contract_advances() -> None: with pytest.raises(SystemExit): _runner_main(["panel", "--contract", "sota-v2", "--dry-run"]) @@ -1602,6 +1648,27 @@ def test_endpoint_preflight_requires_frozen_healthy_capable_route() -> None: assert "cannot honor required parameters" in _endpoint_issues(cell, valid)[0] +def test_endpoint_preflight_allows_explicit_null_cap_deferral_only_until_strict_smoke() -> None: + cell = build_cells("smoke", model_id="openrouter-qwen3.7-plus-alibaba", cap=4096)[0] + payload = {"data": {"endpoints": [_healthy_endpoint(cell)]}} + payload["data"]["endpoints"][0]["max_completion_tokens"] = None + + assert "cannot honor required parameters" in _endpoint_issues(cell, payload)[0] + + deferred = replace( + cell, + output_cap_verification=dict(PENDING_STRICT_SMOKE_CAP_VERIFICATION), + ) + assert _endpoint_issues(deferred, payload) == [] + + payload["data"]["endpoints"][0]["supported_parameters"].remove("max_tokens") + assert "cannot honor required parameters" in _endpoint_issues(deferred, payload)[0] + + payload["data"]["endpoints"][0]["supported_parameters"].append("max_tokens") + payload["data"]["endpoints"][0]["max_completion_tokens"] = 2048 + assert "cannot honor required parameters" in _endpoint_issues(deferred, payload)[0] + + def _healthy_endpoint(cell) -> dict: return { "provider_name": "Alibaba", diff --git a/tests/test_sota_v3_preregistration.py b/tests/test_sota_v3_preregistration.py index 16fdc06..0386f23 100644 --- a/tests/test_sota_v3_preregistration.py +++ b/tests/test_sota_v3_preregistration.py @@ -14,6 +14,7 @@ exact_sign_flip_feasibility, publication_execution_issues, v3_preregistration_coherence_issues, + v3_route_acceptance_issues, ) CONFIG = Path("config") @@ -67,16 +68,14 @@ def test_v3_lane_pins_current_contract_and_freezes_a_powered_allocation() -> Non # The selection rule is a Wilson *lower* bound clearing the target, so the # allocation is only frozen if the conservative-sensitivity interval does. assert selected["sensitivity_power_wilson_ci95"][0] >= candidate["target_familywise_all_reject_power"] - assert lane["seed_panel"] == { - "status": "pending-authorized-generation", - "name": None, - "count": 16, - "sha256": None, - } + assert lane["seed_panel"]["status"] == "frozen" + assert lane["seed_panel"]["name"] == "private-env" + assert lane["seed_panel"]["count"] == 16 + assert len(lane["seed_panel"]["sha256"]) == 64 + assert len(lane["seed_panel"]["hiding_commitment_sha256"]) == 64 + assert lane["seed_panel"]["seed_values_included"] is False assert lane["seed_panel"]["count"] == selected["seed_count"] - # Seed identity stays unfrozen until a salted commitment hash is recorded. - assert lane["seed_panel"]["sha256"] is None - assert any("private panel" in blocker.lower() for blocker in lane["blockers"]) + assert any("strict smoke" in blocker.lower() for blocker in lane["blockers"]) assert lane["reference_agent"] == "pick-trader" assert lane["protocol_repair_attempts"] == 1 assert lane["strict_fallback_required"] is True @@ -87,31 +86,29 @@ def test_v3_lane_pins_current_contract_and_freezes_a_powered_allocation() -> Non assert lane["minimum_headline_models"] >= 8 -def test_v3_registry_is_truthfully_provisional_and_contains_no_unverified_routes() -> None: +def test_v3_registry_is_frozen_to_authenticated_routes_and_privacy_evidence() -> None: lane = _read("sota_v3_lane.json") registry = _read("sota_v3_models.json") assert registry["contract"] == lane["contract"] assert registry["contract_fingerprint"] == lane["contract_fingerprint"] - # route-preflight-ready is strictly weaker than frozen: it clears the - # zero-call preflight readiness check and nothing else. The registry is - # still not frozen, so every paid phase stays locked. - assert registry["selection_status"] == "route-preflight-ready" - assert registry["selection_frozen_at_utc"] is None + assert registry["selection_status"] == "frozen" + assert registry["selection_frozen_at_utc"] assert registry["catalog_snapshot_status"] == "frozen-public-metadata-only" assert registry["catalog_checked_at_utc"] assert len(registry["models"]) == len(registry["required_smokes"]) == 10 assert set(registry["required_smokes"]) == {model["id"] for model in registry["models"]} assert registry["repeats"] == lane["repeats"] == 1 assert registry["output_token_cap"] == lane["output_token_cap"] == 4096 - assert registry["output_budget_status"] == lane["output_budget_status"] == "provisional-pre-smoke-validation" - assert registry["spend_authorized"] is False + assert registry["output_budget_status"] == lane["output_budget_status"] == "frozen-native-reasoning-cap" + assert registry["spend_authorized"] is True assert registry["panel_execution_authorized"] is False - assert registry["unresolved_decisions"] + assert registry["exact_route_acceptance"]["status"] == "accepted" + assert v3_route_acceptance_issues(registry) == [] assert registry["provider_policy"] == "openrouter-only" assert registry["shared_fixed_options"]["OPENROUTER_JSON_MODE"] == "true" assert registry["public_metadata_limitations"] - assert any("privacy" in decision for decision in registry["unresolved_decisions"]) + assert all("smoke" in decision for decision in registry["unresolved_decisions"]) def test_v3_protocol_and_pricing_are_separate_and_fail_closed() -> None: @@ -121,7 +118,7 @@ def test_v3_protocol_and_pricing_are_separate_and_fail_closed() -> None: assert protocol["contract"] == pricing["contract"] == lane["contract"] assert protocol["contract_fingerprint"] == pricing["contract_fingerprint"] == lane["contract_fingerprint"] - assert protocol["status"] == "provisional-pre-smoke" + assert protocol["status"] == "frozen" assert protocol["statistical_analysis_plan"]["status"] == "frozen" assert protocol["statistical_analysis_plan"]["analysis_mode"] == "reference-only" assert protocol["statistical_analysis_plan"]["inference_method"] == "exact-enumeration-sign-flip" @@ -144,36 +141,32 @@ def test_v3_protocol_and_pricing_are_separate_and_fail_closed() -> None: == lane["statistical_panel_design"]["target_effect_score_points"] ) assert protocol["panel_design"]["status"] == lane["panel_design_status"] - assert protocol["budget_policy"]["spend_authorized"] is False - assert pricing["status"] == "catalog-frozen-public-metadata-only" + assert protocol["budget_policy"]["spend_authorized"] is True + assert pricing["status"] == "frozen" assert pricing["checked_at_utc"] assert len(pricing["models"]) == 10 - assert pricing["spend_authorized"] is False + assert pricing["spend_authorized"] is True -def test_v3_preregistration_fails_closed_before_smoke_or_panel_spend() -> None: +def test_v3_preregistration_authorizes_smoke_but_keeps_panel_and_publication_locked() -> None: lane = _read("sota_v3_lane.json") registry = _read("sota_v3_models.json") manifest = _read("sota_v3_smoke_manifest.json") - # The design amendment freezes an allocation; it authorizes nothing. Both - # gates are allowlists against the literal "frozen", so a status that merely - # records progress still locks provider execution. - assert lane["preregistration_status"] == "provisional-pre-smoke" - assert lane["preregistration_status"] != "frozen" + assert lane["preregistration_status"] == "frozen" assert lane["panel_design_status"] == "frozen" - assert lane["output_budget_status"] == "provisional-pre-smoke-validation" + assert lane["output_budget_status"] == "frozen-native-reasoning-cap" assert lane["output_token_cap"] == 4096 assert lane["cap_pressure_threshold_tokens"] == 3072 assert lane["fallback_output_token_cap"] == 8192 assert "invalidate every v3 smoke" in lane["output_policy_amendment_rule"] - assert lane["reasoning_policy"] == "catalog-pinned-pending-live-route-verification" - assert lane["spend_authorized"] is False + assert lane["reasoning_policy"] == "catalog-pinned-pending-strict-smoke-behavior-verification" + assert lane["spend_authorized"] is True # Granted 2026-08-03 for the completed zero-completion-call probe; it is the # one authorization that cannot reach a model, reserve spend, or write run # state, so it does not belong in the fail-closed set below. assert lane["route_preflight_authorized"] is True - assert lane["smoke_execution_authorized"] is False + assert lane["smoke_execution_authorized"] is True assert lane["panel_execution_authorized"] is False assert lane["publication_authorized"] is False assert lane["blockers"] @@ -225,7 +218,7 @@ def test_exact_sign_flip_holm_feasibility_uses_seed_count_not_episode_count() -> assert nine_seeds["feasible"] is True -def test_blocked_v3_state_cannot_drift_into_partial_authorization() -> None: +def test_smoke_authorization_cannot_drift_into_panel_or_publication_authorization() -> None: lane = _read("sota_v3_lane.json") registry = _read("sota_v3_models.json") manifest = _read("sota_v3_smoke_manifest.json") @@ -238,20 +231,15 @@ def test_blocked_v3_state_cannot_drift_into_partial_authorization() -> None: or set(registry["required_smokes"]) != set(manifest["entries"]) ) assert incomplete - assert not any( - ( - lane["spend_authorized"], - lane["smoke_execution_authorized"], - lane["panel_execution_authorized"], - lane["publication_authorized"], - registry["spend_authorized"], - registry["panel_execution_authorized"], - ) - ) + assert lane["spend_authorized"] is lane["smoke_execution_authorized"] is True + assert registry["spend_authorized"] is True + assert lane["panel_execution_authorized"] is False + assert lane["publication_authorized"] is False + assert registry["panel_execution_authorized"] is False @pytest.mark.parametrize("mode", ["--dry-run", "--preflight-only"]) -def test_runner_rejects_provisional_v3_smoke_before_provider_access( +def test_runner_requires_private_seed_escrow_before_provider_access( monkeypatch: pytest.MonkeyPatch, capsys: pytest.CaptureFixture[str], mode: str, @@ -272,7 +260,7 @@ def test_runner_rejects_provisional_v3_smoke_before_provider_access( publication_runner.main(["smoke", "--contract", "sota-v3", mode]) assert exc_info.value.code == 2 - assert "sota-v3 lane is provisional-pre-smoke; provider execution is locked" in capsys.readouterr().err + assert "GM_BENCH_PRIVATE_SEEDS is required for the frozen private panel" in capsys.readouterr().err assert provider_access == [] @@ -334,14 +322,7 @@ def test_runner_requires_explicit_contract_before_v2_preflight_can_run( assert provider_access == [] -def test_route_preflight_readiness_unlocks_nothing_that_costs_money() -> None: - """`route-preflight-ready` must buy exactly one thing: the zero-call probe. - - This is the whole safety argument for the flip, so it is asserted rather - than described. The registry is deliberately *not* `frozen`; every paid - phase must remain as locked as it was while the registry was - `provisional-blocked`, both before and after the probe actually ran. - """ +def test_smoke_readiness_unlocks_smoke_but_not_panel() -> None: lane = _read("sota_v3_lane.json") registry = _read("sota_v3_models.json") protocol = _read("sota_v3_publication_protocol.json") @@ -351,7 +332,6 @@ def test_route_preflight_readiness_unlocks_nothing_that_costs_money() -> None: def issues(reg: dict, phase: str) -> list[str]: return publication_execution_issues(lane, reg, manifest, phase=phase, protocol=protocol, pricing=pricing) - # The probe is authorized and clear; that buys nothing downstream. assert issues(registry, "route-preflight") == [] locked_lane = dict(lane, route_preflight_authorized=False) assert publication_execution_issues( @@ -363,7 +343,10 @@ def issues(reg: dict, phase: str) -> list[str]: pricing=pricing, ) == ["zero-call route preflight is locked while route_preflight_authorized is false"] + assert issues(registry, "smoke") == [] + panel_issues = issues(registry, "panel") + assert "panel execution is locked by the model registry" in panel_issues + assert "v3 smoke manifest is not accepted for panel execution" in panel_issues + blocked = dict(registry, selection_status="provisional-blocked") - for phase in ("smoke", "panel"): - assert issues(registry, phase) == issues(blocked, phase) - assert "provider execution is locked until the model registry is frozen" in issues(registry, phase) + assert "provider execution is locked until the model registry is frozen" in issues(blocked, "smoke") diff --git a/tests/test_sota_v3_route_catalog.py b/tests/test_sota_v3_route_catalog.py index 8270bfd..b47c6e4 100644 --- a/tests/test_sota_v3_route_catalog.py +++ b/tests/test_sota_v3_route_catalog.py @@ -6,7 +6,7 @@ from pathlib import Path import scripts.run_publication_matrix as publication_runner -from gm_bench.publication import publication_execution_issues +from gm_bench.publication import canonical_sha256, publication_execution_issues, v3_route_acceptance_issues CONFIG = Path("config") @@ -42,40 +42,34 @@ def test_v3_catalog_freezes_exact_balanced_cohort_without_unlocking_execution() assert "meta/muse-spark-1.1" not in identities assert registry["catalog_snapshot_status"] == "frozen-public-metadata-only" - assert registry["selection_status"] == "route-preflight-ready" - assert registry["selection_frozen_at_utc"] is None + assert registry["selection_status"] == "frozen" + assert registry["selection_frozen_at_utc"] assert registry["selection_revision"] == "2026-08-04-public-catalog-lineup-refresh-v3" policy = registry["selection_policy"] assert "Qwen 3.8 Max" in policy assert "deepinfra/fp8" in policy and "cloudflare/fp8" in policy - assert "route-preflight-ready rather than frozen" in policy + assert "now frozen" in policy assert registry["catalog_checked_at_utc"] assert set(registry["required_smokes"]) == {model["id"] for model in models} assert registry["output_token_cap"] == 4_096 - assert registry["output_budget_status"] == "provisional-pre-smoke-validation" + assert registry["output_budget_status"] == "frozen-native-reasoning-cap" acceptance = registry["exact_route_acceptance"] - assert acceptance["status"] == "unresolved" + assert acceptance["status"] == "accepted" + assert acceptance["privacy_standard"]["provider_data_collection"] == "deny" + assert acceptance["privacy_standard"]["provider_training_use_allowed"] is False + assert acceptance["privacy_standard"]["zero_data_retention_required"] is False assert set(acceptance["entries"]) == {model["id"] for model in models} for entry in acceptance["entries"].values(): - assert entry["authenticated"] is False - assert entry["route_identity_sha256"] is None - assert entry["privacy_acceptance"]["status"] == "unresolved" - assert not any( - entry["privacy_acceptance"][field] - for field in ( - "data_collection_policy_accepted", - "retention_policy_accepted", - "training_use_policy_accepted", - "zero_data_retention_policy_accepted", - ) - ) - for key in ( - "spend_authorized", - "route_preflight_authorized", - "panel_execution_authorized", - "publication_authorized", - ): - assert registry[key] is False, key + assert entry["authenticated"] is True + assert len(entry["route_identity_sha256"]) == 64 + privacy = entry["privacy_acceptance"] + assert privacy["status"] == "accepted" + assert privacy["zero_data_retention_requirement_satisfied"] is True + assert isinstance(privacy["zero_data_retention_endpoint"], bool) + assert registry["spend_authorized"] is True + assert registry["route_preflight_authorized"] is True + assert registry["panel_execution_authorized"] is False + assert registry["publication_authorized"] is False def test_v3_catalog_pins_routes_parameters_reasoning_and_exact_route_prices() -> None: @@ -112,11 +106,11 @@ def test_v3_catalog_pins_routes_parameters_reasoning_and_exact_route_prices() -> grok = next(model for model in registry["models"] if model["model"] == "x-ai/grok-4.5") assert grok["endpoint_tag"] == "xai/zdr" - assert pricing["status"] == "catalog-frozen-public-metadata-only" + assert pricing["status"] == "frozen" assert pricing["checked_at_utc"] == registry["catalog_checked_at_utc"] - assert pricing["spend_authorized"] is False - assert pricing["route_preflight_authorized"] is False - assert pricing["smoke_execution_authorized"] is False + assert pricing["spend_authorized"] is True + assert pricing["route_preflight_authorized"] is True + assert pricing["smoke_execution_authorized"] is True assert pricing["panel_execution_authorized"] is False assert pricing["publication_authorized"] is False assert pricing["runtime_observations"]["source"] is None @@ -140,45 +134,67 @@ def test_selected_catalog_models_match_runner_exact_route_shape() -> None: ) -def test_completed_route_preflight_cannot_unlock_any_paid_phase() -> None: - """The zero-call phase is open; every phase that spends money is not. +def test_v3_route_acceptance_is_bound_to_public_zero_completion_evidence() -> None: + registry = _read("sota_v3_models.json") + acceptance = registry["exact_route_acceptance"] + evidence = json.loads(Path(acceptance["evidence_artifact"]).read_text()) + + assert evidence["format"] == "gm-bench-route-acceptance-evidence-v1" + assert evidence["completion_calls"] == 0 + assert evidence["account_authentication"]["sensitive_values_included"] is False + assert set(evidence["routes"]) == set(acceptance["entries"]) + assert sum(route["zero_data_retention_endpoint"] for route in evidence["routes"].values()) == 5 + for model_id, route in evidence["routes"].items(): + entry = acceptance["entries"][model_id] + assert entry["route_evidence_sha256"] == canonical_sha256(route) + privacy_evidence = { + "route_identity_sha256": route["route_identity_sha256"], + "privacy_standard": evidence["privacy_standard"], + "zero_data_retention_endpoint": route["zero_data_retention_endpoint"], + "provider_policy": route["provider_policy"], + "official_policy_sources": evidence["official_policy_sources"], + } + assert entry["privacy_acceptance"]["evidence_sha256"] == canonical_sha256(privacy_evidence) + + assert v3_route_acceptance_issues(registry) == [] + first_entry = registry["exact_route_acceptance"]["entries"][next(iter(evidence["routes"]))] + route_sha = first_entry["route_evidence_sha256"] + first_entry["route_evidence_sha256"] = "0" * 64 + assert any("route evidence digest does not match" in issue for issue in v3_route_acceptance_issues(registry)) + first_entry["route_evidence_sha256"] = route_sha + first_entry["privacy_acceptance"]["evidence_sha256"] = "0" * 64 + assert any("privacy evidence digest does not match" in issue for issue in v3_route_acceptance_issues(registry)) + - `route_preflight_authorized` was granted on 2026-08-03 and the preflight - has run, so this no longer asserts that *nothing* is unlocked. It asserts - the distinction the lane is built on: opening the zero-completion-call - probe must leave the paid gates exactly where they were. - """ +def test_smoke_authorization_still_cannot_unlock_panel_or_publication() -> None: lane = _read("sota_v3_lane.json") registry = _read("sota_v3_models.json") protocol = _read("sota_v3_publication_protocol.json") pricing = _read("sota_v3_pricing_snapshot.json") manifest = _read("sota_v3_smoke_manifest.json") - # Granted: the phase that provably cannot call a model or reserve spend. assert lane["route_preflight_authorized"] is True - # Not granted: everything that can. - assert lane["spend_authorized"] is False - assert lane["smoke_execution_authorized"] is False + assert lane["spend_authorized"] is True + assert lane["smoke_execution_authorized"] is True assert lane["panel_execution_authorized"] is False assert lane["publication_authorized"] is False - assert protocol["budget_policy"]["spend_authorized"] is False + assert protocol["budget_policy"]["spend_authorized"] is True assert protocol["publication_authorized"] is False assert manifest["accepted_for_panel"] is False - assert pricing["route_preflight_authorized"] is False + assert pricing["route_preflight_authorized"] is True - for phase in ("smoke", "panel"): - issues = publication_execution_issues( + assert ( + publication_execution_issues( lane, registry, manifest, - phase=phase, + phase="smoke", protocol=protocol, pricing=pricing, ) - assert issues, f"completed route preflight unexpectedly unlocked {phase}" + == [] + ) - # Preflight is deliberately clear now; a passing probe is not a licence to - # spend, so the registry must still be unfrozen and acceptance unresolved. assert ( publication_execution_issues( lane, @@ -190,17 +206,16 @@ def test_completed_route_preflight_cannot_unlock_any_paid_phase() -> None: ) == [] ) - assert registry["selection_status"] == "route-preflight-ready" - assert registry["selection_frozen_at_utc"] is None + assert registry["selection_status"] == "frozen" + assert registry["selection_frozen_at_utc"] - smoke_issues = publication_execution_issues( + panel_issues = publication_execution_issues( lane, registry, manifest, - phase="smoke", + phase="panel", protocol=protocol, pricing=pricing, ) - assert "sota-v3 exact-route acceptance status is not accepted" in smoke_issues - assert any("lacks authenticated route verification" in issue for issue in smoke_issues) - assert any("privacy acceptance is unresolved" in issue for issue in smoke_issues) + assert "panel execution is locked by the model registry" in panel_issues + assert "v3 smoke manifest is not accepted for panel execution" in panel_issues diff --git a/tests/test_sota_v3_route_evidence.py b/tests/test_sota_v3_route_evidence.py new file mode 100644 index 0000000..eabd658 --- /dev/null +++ b/tests/test_sota_v3_route_evidence.py @@ -0,0 +1,45 @@ +from __future__ import annotations + +from pathlib import Path + +import pytest + +import scripts.collect_sota_v3_route_evidence as collector + + +def test_route_evidence_http_client_rejects_non_api_paths(monkeypatch: pytest.MonkeyPatch) -> None: + accessed: list[str] = [] + monkeypatch.setattr( + collector.http.client, + "HTTPSConnection", + lambda *_args, **_kwargs: accessed.append("called"), + ) + + for path in ( + "file:///etc/passwd", + "https://example.com/api/v1/providers", + "//example.com/api/v1/providers", + "/not-api/providers", + "/api/v1/providers?unexpected=query", + ): + with pytest.raises(ValueError, match="refusing non-OpenRouter metadata path"): + collector._get_json(path, {}) + assert accessed == [] + + +@pytest.mark.parametrize("option", ["--registry", "--output"]) +def test_route_evidence_cli_rejects_paths_outside_repository( + monkeypatch: pytest.MonkeyPatch, + tmp_path: Path, + option: str, +) -> None: + accessed: list[str] = [] + monkeypatch.setattr( + collector.http.client, + "HTTPSConnection", + lambda *_args, **_kwargs: accessed.append("called"), + ) + + with pytest.raises(SystemExit): + collector.main([option, str(tmp_path / "outside.json")]) + assert accessed == [] diff --git a/tests/test_sota_v3_smoke_keychain.py b/tests/test_sota_v3_smoke_keychain.py new file mode 100644 index 0000000..d5b4636 --- /dev/null +++ b/tests/test_sota_v3_smoke_keychain.py @@ -0,0 +1,63 @@ +from __future__ import annotations + +import json +from types import SimpleNamespace + +import pytest + +import scripts.run_sota_v3_smoke_from_keychain as launcher +from gm_bench.benchmark_config import PRIVATE_SEEDS_ENV, seed_panel_hash +from scripts.seed_panel_commitment import commitment + + +def _install_keychain_fixture(tmp_path, monkeypatch: pytest.MonkeyPatch) -> str: + seeds = [(1 << 50) + index * 104729 for index in range(16)] + seeds_text = ",".join(str(seed) for seed in seeds) + salt = "ab" * 32 + record = { + "format": "gm-bench-private-seed-secret-v1", + "seeds": seeds_text, + "salt": salt, + } + config = tmp_path / "config" + config.mkdir() + (config / "sota_v3_lane.json").write_text( + json.dumps( + { + "seed_panel": { + "status": "frozen", + "name": "private-env", + "count": len(seeds), + "sha256": seed_panel_hash(seeds), + "hiding_commitment_sha256": commitment(salt, seeds), + } + } + ) + ) + monkeypatch.setattr(launcher, "ROOT", tmp_path) + monkeypatch.setattr( + launcher.subprocess, + "run", + lambda *_args, **_kwargs: SimpleNamespace(stdout=json.dumps(record).encode().hex() + "\n"), + ) + return seeds_text + + +def test_keychain_launcher_verifies_hex_encoded_private_panel(tmp_path, monkeypatch: pytest.MonkeyPatch) -> None: + expected = _install_keychain_fixture(tmp_path, monkeypatch) + assert launcher._verified_seed_text() == expected + + +def test_keychain_launcher_sets_seed_env_only_for_runner(tmp_path, monkeypatch: pytest.MonkeyPatch) -> None: + expected = _install_keychain_fixture(tmp_path, monkeypatch) + observed: list[tuple[list[str], str | None]] = [] + monkeypatch.setattr( + launcher, + "publication_main", + lambda argv: observed.append((argv, launcher.os.environ.get(PRIVATE_SEEDS_ENV))) or 0, + ) + + assert launcher.main(["--max-spend-usd", "150", "--dry-run"]) == 0 + assert observed[0][1] == expected + assert PRIVATE_SEEDS_ENV not in launcher.os.environ + assert observed[0][0][:3] == ["smoke", "--contract", "sota-v3"]