diff --git a/config/sota_v3_lane.json b/config/sota_v3_lane.json index 5998f9d..c927f22 100644 --- a/config/sota_v3_lane.json +++ b/config/sota_v3_lane.json @@ -6,23 +6,25 @@ "mechanics_change_policy": "Any score-, action-, observation-, or simulator-semantic change after this preregistration requires a new contract fingerprint, invalidates all sota-v3 smoke evidence, and re-locks provider spend.", "preregistration_status": "provisional-pre-smoke", "preregistered_at_utc": "2026-07-27T16:58:41Z", - "design_amended_at_utc": "2026-07-28T00:00:00Z", + "design_amended_at_utc": "2026-08-03T00:00:00Z", "design_amendment": { - "amendment_id": "sota-v3-design-amendment-1", + "amendment_id": "sota-v3-design-amendment-2", "status": "pre-data", - "supersedes": "docs/run_logs/sota-v3-statistical-design-audit-2026-07-28.md", - "record": "docs/run_logs/sota-v3-design-amendment-2026-07-28.md", - "rationale": "The superseded design powered a superiority claim (+40 above pick-trader) that the frozen sota-v2 evidence contradicts: all eight eligible models trailed the reference by 180-282 score points with a 0.0 seed win rate on every model. No allocation in the 9-20 seed by 1-3 repeat grid reached 0.80 sensitivity power for that claim, and the 20-seed ceiling caps sensitivity power near 0.36 even at unbounded repeats, so the block was structural rather than a budget shortfall.", + "supersedes": "docs/run_logs/sota-v3-design-amendment-2026-07-28.md", + "record": "docs/run_logs/sota-v3-design-amendment-2026-08-03.md", + "rationale": "Owner-directed pre-data cohort update before any smoke or panel evidence exists: the OpenAI anchor moves from GPT-5.6 Luna Pro to plain GPT-5.6 Luna (the originally intended route, now healthy, without the Pro variant's reasoning.mode inconsistency), and DeepSeek V4 Flash 0731 plus Tencent Hy3 join as open-weight anchors on first-party FP8 routes; Thinking Machines Inkling Small was evaluated and found ineligible because no healthy route advertises response_format under the lane's frozen JSON-mode and require-parameters options. A family of ten tightens the Holm first step from 0.00625 to 0.005, so the allocation was reselected with the identical frozen power machinery; 15x1 no longer holds the 0.80 sensitivity floor at family ten and the smallest qualifying allocation is 16 seeds x 1 repeat.", "changes": [ - "Primary claim restated directionally: each registered model trails pick-trader, replicating the frozen sota-v2 finding under the sota-v3 contract.", - "Planning effect changed from +40 to -100 score points, conservative against the -180 to -282 observed on sota-v2.", - "Panel repeats changed from 3 to 1: at a fixed episode budget the candidate-minus-reference lift keeps its full seed component, so repeats shrink only the within-seed noise term and seeds dominate for discrimination.", - "Evaluated seed grid narrowed from 9-20 to 9-16, since the qualifying allocation falls inside it." + "Registered model family grows from eight to ten; Holm family size is now 10 and the first-step threshold is 0.05/10 = 0.005.", + "GPT-5.6 Luna Pro is replaced by plain GPT-5.6 Luna on the first-party OpenAI route, removing the unresolved Pro reasoning.mode inconsistency.", + "DeepSeek V4 Flash 0731 and Tencent Hy3 are added as open-weight anchors, both on first-party FP8 routes.", + "Selected allocation moves from 15 seeds x 1 repeat to 16 seeds x 1 repeat (16 episodes/model): sensitivity power 0.8488, Wilson 95% CI [0.841645, 0.855688]; base power 0.9527, Wilson 95% CI [0.948363, 0.95669].", + "The pending private seed panel is now 16 seeds; its identity remains unfrozen pending separately authorized generation." ], "unchanged": [ - "alpha 0.05, Holm-Bonferroni across a fixed family of eight", + "alpha 0.05, Holm-Bonferroni across the fixed registered family", "exact two-sided enumeration sign-flip test with the seed as the unit of inference", - "0.80 conservative-sensitivity familywise all-reject power target", + "planning effect -100 score points and the 0.80 conservative-sensitivity familywise all-reject power target", + "historical variance components, sensitivity multipliers, 10000 trials, simulation seed 2026072800", "no model-to-model tiers or ordinal ranking", "every execution, spend, and publication authorization remains false" ] @@ -40,9 +42,15 @@ "status": "frozen", "claim_direction": "trails-reference", "primary_claim": "Every registered model-plus-compact-scaffold system trails deterministic pick-trader on seed-paired mean lift under the frozen sota-v3 mechanics.", - "evaluated_seed_range": [9, 16], - "evaluated_repeat_range": [1, 3], - "holm_family_size": 8, + "evaluated_seed_range": [ + 9, + 16 + ], + "evaluated_repeat_range": [ + 1, + 3 + ], + "holm_family_size": 10, "target_effect_score_points": -100, "target_effect_basis": "Conservative against the frozen sota-v2 observed lifts of -180 to -282 score points at a 0.0 seed win rate for all eight eligible models.", "target_familywise_all_reject_power": 0.8, @@ -60,16 +68,22 @@ "interaction_variance_floor_as_historical_noise_fraction": 0.1 }, "selected_allocation": { - "seed_count": 15, + "seed_count": 16, "repeats": 1, - "episodes_per_model": 15, - "minimum_exact_two_sided_sign_flip_p_value": 6.103515625e-05, - "holm_first_step_threshold_at_alpha_0_05": 0.00625, + "episodes_per_model": 16, + "minimum_exact_two_sided_sign_flip_p_value": 3.0517578125e-05, + "holm_first_step_threshold_at_alpha_0_05": 0.005, "exact_sign_flip_holm_feasible": true, - "base_power_estimate": 0.9461, - "base_power_wilson_ci95": [0.9415, 0.950357], - "sensitivity_power_estimate": 0.8357, - "sensitivity_power_wilson_ci95": [0.828309, 0.842834] + "base_power_estimate": 0.9527, + "base_power_wilson_ci95": [ + 0.948363, + 0.95669 + ], + "sensitivity_power_estimate": 0.8488, + "sensitivity_power_wilson_ci95": [ + 0.841645, + 0.855688 + ] }, "selection_rule": "Smallest allocation with 9-16 seeds and 1-3 repeats whose conservative-sensitivity familywise-power Wilson 95% lower bound is at least 0.80.", "decision_required": null @@ -77,7 +91,7 @@ "seed_panel": { "status": "pending-authorized-generation", "name": null, - "count": 15, + "count": 16, "sha256": null }, "reference_agent": "pick-trader", @@ -87,7 +101,7 @@ "smoke_manifest": "config/sota_v3_smoke_manifest.json", "publication_protocol": "config/sota_v3_publication_protocol.json", "pricing_snapshot": "config/sota_v3_pricing_snapshot.json", - "minimum_headline_models": 8, + "minimum_headline_models": 10, "reasoning_policy": "catalog-pinned-pending-live-route-verification", "output_token_cap": 4096, "output_budget_status": "provisional-pre-smoke-validation", @@ -101,7 +115,7 @@ "panel_execution_authorized": false, "publication_authorized": false, "blockers": [ - "Generate and commit the 15-seed private panel under separate owner authorization; seed identity is unfrozen until its salted commitment hash is recorded.", + "Generate and commit the 16-seed private panel under separate owner authorization; seed identity is unfrozen until its salted commitment hash is recorded.", "Complete authenticated exact-route and privacy verification for the catalog-selected model cohort before any provider call.", "Live-verify and pin each exact provider route, endpoint name, supported parameters, privacy policy, and reasoning requirements.", "Validate the provisional common 4,096-token safety ceiling on one strict smoke per registered route; any predeclared cap-pressure trigger invalidates all v3 smokes and requires one symmetric pre-panel amendment.", diff --git a/config/sota_v3_models.json b/config/sota_v3_models.json index 75c6e5b..fa58409 100644 --- a/config/sota_v3_models.json +++ b/config/sota_v3_models.json @@ -11,41 +11,61 @@ "repeats": 1, "selection_status": "provisional-blocked", "selection_frozen_at_utc": null, - "selection_revision": "2026-07-28-public-catalog-cohort-v1", + "selection_revision": "2026-08-03-public-catalog-cohort-v2", "catalog_snapshot_status": "frozen-public-metadata-only", - "catalog_checked_at_utc": "2026-07-28T04:15:49Z", + "catalog_checked_at_utc": "2026-08-03T15:53:59Z", "catalog_sources": [ "https://openrouter.ai/api/v1/models", "https://openrouter.ai/api/v1/models/{model_id}/endpoints" ], - "selection_policy": "Eight-model pre-data cohort selected from the 2026-07-28 public OpenRouter catalog. GPT-5.6 Luna Pro preserves the prior audit's pre-data substitution for the then-unhealthy Luna route even though Luna's public status had recovered by this snapshot, and Gemini 3.6 Flash replaces Gemini 3.5 Flash. Exact public endpoint metadata and prices are pinned below, but the registry remains provisional-blocked because public metadata does not prove authenticated exact-route access or provider privacy and retention behavior.", + "selection_policy": "Ten-model pre-data cohort selected from the 2026-08-03 public OpenRouter catalog, superseding the eight-model 2026-07-28 cohort before any smoke or panel evidence exists. GPT-5.6 Luna replaces GPT-5.6 Luna Pro: the plain Luna route that was unhealthy at the prior snapshot has recovered, and the Pro variant carried an unresolved reasoning.mode inconsistency between its public description and the structured catalog. DeepSeek V4 Flash 0731 and Tencent Hy3 (both on first-party FP8 routes) are added as open-weight anchors. Thinking Machines Inkling Small was evaluated for the tenth slot and found ineligible at this snapshot: no healthy route advertises the response_format parameter the lane's frozen JSON-mode and require-parameters options demand. Exact public endpoint metadata and prices are pinned below, but the registry remains provisional-blocked because public metadata does not prove authenticated exact-route access or provider privacy and retention behavior.", "models": [ { - "id": "openrouter-gpt-5.6-luna-pro-openai", + "id": "openrouter-gpt-5.6-luna-openai", "provider": "openrouter", - "model": "openai/gpt-5.6-luna-pro", - "canonical_slug": "openai/gpt-5.6-luna-pro-20260709", + "model": "openai/gpt-5.6-luna", + "canonical_slug": "openai/gpt-5.6-luna-20260709", "transport": "gateway-api", "cohort": "frontier-proprietary", "upstream_provider": "OpenAI", "upstream_provider_slug": "openai", "endpoint_tag": "openai", - "endpoint_name": "OpenAI | openai/gpt-5.6-luna-pro-20260709", + "endpoint_name": "OpenAI | openai/gpt-5.6-luna-20260709", "catalog_route_status": 0, - "catalog_uptime_last_30m": 100.0, - "catalog_supported_parameters": ["reasoning", "include_reasoning", "seed", "max_tokens", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort"], + "catalog_uptime_last_30m": 99.31771054990628, + "catalog_supported_parameters": [ + "reasoning", + "include_reasoning", + "seed", + "max_tokens", + "response_format", + "structured_outputs", + "tools", + "tool_choice", + "reasoning_effort" + ], "reasoning_policy": "disabled", "reasoning_effort": null, "catalog_reasoning": { "mandatory": false, "default_enabled": true, - "supported_efforts": ["max", "xhigh", "high", "medium", "low", "none"], - "default_effort": "medium", - "note": "The public description says the Pro variant is served with reasoning.mode=pro, while the structured catalog marks reasoning optional and supports none. A strict smoke is required to resolve that inconsistency before evidence can be accepted." + "supported_efforts": [ + "max", + "xhigh", + "high", + "medium", + "low", + "none" + ], + "default_effort": "medium" + }, + "fixed_options": { + "OPENROUTER_REASONING_ENABLED": "false" }, - "fixed_options": {"OPENROUTER_REASONING_ENABLED": "false"}, - "absent_options": ["OPENROUTER_REASONING_EFFORT"], - "role": "OpenAI frontier anchor; preserves the pre-data substitution made when the non-Pro Luna route was unhealthy" + "absent_options": [ + "OPENROUTER_REASONING_EFFORT" + ], + "role": "OpenAI frontier anchor; restores the originally intended non-Pro Luna route now that its catalog status has recovered, removing the Pro variant's reasoning.mode inconsistency" }, { "id": "openrouter-claude-sonnet-5-bedrock", @@ -60,17 +80,38 @@ "endpoint_name": "Amazon Bedrock | anthropic/claude-sonnet-5-20260630", "catalog_route_status": 0, "catalog_uptime_last_30m": 99.56639566395664, - "catalog_supported_parameters": ["reasoning", "include_reasoning", "max_tokens", "stop", "tools", "tool_choice", "structured_outputs", "response_format", "verbosity", "reasoning_effort"], + "catalog_supported_parameters": [ + "reasoning", + "include_reasoning", + "max_tokens", + "stop", + "tools", + "tool_choice", + "structured_outputs", + "response_format", + "verbosity", + "reasoning_effort" + ], "reasoning_policy": "disabled", "reasoning_effort": null, "catalog_reasoning": { "mandatory": false, "default_enabled": true, - "supported_efforts": ["max", "xhigh", "high", "medium", "low"], + "supported_efforts": [ + "max", + "xhigh", + "high", + "medium", + "low" + ], "default_effort": "high" }, - "fixed_options": {"OPENROUTER_REASONING_ENABLED": "false"}, - "absent_options": ["OPENROUTER_REASONING_EFFORT"], + "fixed_options": { + "OPENROUTER_REASONING_ENABLED": "false" + }, + "absent_options": [ + "OPENROUTER_REASONING_EFFORT" + ], "role": "Anthropic frontier anchor on the previously selected global Bedrock route" }, { @@ -86,16 +127,36 @@ "endpoint_name": "Google AI Studio | google/gemini-3.6-flash-20260721", "catalog_route_status": 0, "catalog_uptime_last_30m": 99.78925184404636, - "catalog_supported_parameters": ["reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "response_format", "structured_outputs", "tool_choice", "tools", "reasoning_effort"], + "catalog_supported_parameters": [ + "reasoning", + "include_reasoning", + "max_tokens", + "temperature", + "top_p", + "seed", + "response_format", + "structured_outputs", + "tool_choice", + "tools", + "reasoning_effort" + ], "reasoning_policy": "mandatory-minimum", "reasoning_effort": "minimal", "catalog_reasoning": { "mandatory": true, "default_enabled": true, - "supported_efforts": ["high", "medium", "low", "minimal"], + "supported_efforts": [ + "high", + "medium", + "low", + "minimal" + ], "default_effort": "medium" }, - "fixed_options": {"OPENROUTER_REASONING_ENABLED": "true", "OPENROUTER_REASONING_EFFORT": "minimal"}, + "fixed_options": { + "OPENROUTER_REASONING_ENABLED": "true", + "OPENROUTER_REASONING_EFFORT": "minimal" + }, "absent_options": [], "role": "Preferred Google fast frontier anchor" }, @@ -112,16 +173,40 @@ "endpoint_name": "xAI | x-ai/grok-4.5-20260708", "catalog_route_status": 0, "catalog_uptime_last_30m": 100.0, - "catalog_supported_parameters": ["reasoning", "include_reasoning", "structured_outputs", "response_format", "max_tokens", "temperature", "top_p", "seed", "logprobs", "top_logprobs", "stop", "frequency_penalty", "presence_penalty", "tools", "tool_choice", "reasoning_effort"], + "catalog_supported_parameters": [ + "reasoning", + "include_reasoning", + "structured_outputs", + "response_format", + "max_tokens", + "temperature", + "top_p", + "seed", + "logprobs", + "top_logprobs", + "stop", + "frequency_penalty", + "presence_penalty", + "tools", + "tool_choice", + "reasoning_effort" + ], "reasoning_policy": "mandatory-minimum", "reasoning_effort": "low", "catalog_reasoning": { "mandatory": true, "default_enabled": true, - "supported_efforts": ["high", "medium", "low"], + "supported_efforts": [ + "high", + "medium", + "low" + ], "default_effort": "high" }, - "fixed_options": {"OPENROUTER_REASONING_ENABLED": "true", "OPENROUTER_REASONING_EFFORT": "low"}, + "fixed_options": { + "OPENROUTER_REASONING_ENABLED": "true", + "OPENROUTER_REASONING_EFFORT": "low" + }, "absent_options": [], "role": "xAI frontier anchor on the public ZDR-tagged route; the tag is directionally aligned with the privacy goal but is not accepted as retention-policy proof" }, @@ -138,17 +223,40 @@ "endpoint_name": "Novita | z-ai/glm-5.2-20260616", "catalog_route_status": 0, "catalog_uptime_last_30m": 99.70362506358791, - "catalog_supported_parameters": ["reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "top_k", "repetition_penalty", "tools", "tool_choice", "response_format", "reasoning_effort"], + "catalog_supported_parameters": [ + "reasoning", + "include_reasoning", + "max_tokens", + "temperature", + "top_p", + "stop", + "frequency_penalty", + "presence_penalty", + "seed", + "top_k", + "repetition_penalty", + "tools", + "tool_choice", + "response_format", + "reasoning_effort" + ], "reasoning_policy": "disabled", "reasoning_effort": null, "catalog_reasoning": { "mandatory": false, "default_enabled": true, - "supported_efforts": ["xhigh", "high"], + "supported_efforts": [ + "xhigh", + "high" + ], "default_effort": "high" }, - "fixed_options": {"OPENROUTER_REASONING_ENABLED": "false"}, - "absent_options": ["OPENROUTER_REASONING_EFFORT"], + "fixed_options": { + "OPENROUTER_REASONING_ENABLED": "false" + }, + "absent_options": [ + "OPENROUTER_REASONING_EFFORT" + ], "role": "Z.ai open-weight anchor on the previously smoke-validated Novita FP8 route" }, { @@ -164,12 +272,27 @@ "endpoint_name": "Minimax | minimax/minimax-m3-20260531", "catalog_route_status": 0, "catalog_uptime_last_30m": 99.45594322885867, - "catalog_supported_parameters": ["reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "response_format", "tool_choice", "tools"], + "catalog_supported_parameters": [ + "reasoning", + "include_reasoning", + "max_tokens", + "temperature", + "top_p", + "response_format", + "tool_choice", + "tools" + ], "reasoning_policy": "disabled", "reasoning_effort": null, - "catalog_reasoning": {"mandatory": false}, - "fixed_options": {"OPENROUTER_REASONING_ENABLED": "false"}, - "absent_options": ["OPENROUTER_REASONING_EFFORT"], + "catalog_reasoning": { + "mandatory": false + }, + "fixed_options": { + "OPENROUTER_REASONING_ENABLED": "false" + }, + "absent_options": [ + "OPENROUTER_REASONING_EFFORT" + ], "role": "MiniMax open-weight anchor on the first-party FP8 route" }, { @@ -185,12 +308,33 @@ "endpoint_name": "Alibaba | qwen/qwen3.7-plus-20260602", "catalog_route_status": 0, "catalog_uptime_last_30m": 99.99335742374322, - "catalog_supported_parameters": ["reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "seed", "presence_penalty", "response_format", "logprobs", "top_logprobs", "tools", "tool_choice", "structured_outputs"], + "catalog_supported_parameters": [ + "reasoning", + "include_reasoning", + "max_tokens", + "temperature", + "top_p", + "seed", + "presence_penalty", + "response_format", + "logprobs", + "top_logprobs", + "tools", + "tool_choice", + "structured_outputs" + ], "reasoning_policy": "disabled", "reasoning_effort": null, - "catalog_reasoning": {"mandatory": false, "default_enabled": true}, - "fixed_options": {"OPENROUTER_REASONING_ENABLED": "false"}, - "absent_options": ["OPENROUTER_REASONING_EFFORT"], + "catalog_reasoning": { + "mandatory": false, + "default_enabled": true + }, + "fixed_options": { + "OPENROUTER_REASONING_ENABLED": "false" + }, + "absent_options": [ + "OPENROUTER_REASONING_EFFORT" + ], "role": "Qwen open-weight frontier anchor" }, { @@ -206,24 +350,140 @@ "endpoint_name": "Mistral | mistralai/mistral-medium-3.5-20260430", "catalog_route_status": 0, "catalog_uptime_last_30m": 100.0, - "catalog_supported_parameters": ["reasoning", "include_reasoning", "max_tokens", "temperature", "top_p", "stop", "frequency_penalty", "presence_penalty", "seed", "response_format", "structured_outputs", "tools", "tool_choice", "reasoning_effort"], + "catalog_supported_parameters": [ + "reasoning", + "include_reasoning", + "max_tokens", + "temperature", + "top_p", + "stop", + "frequency_penalty", + "presence_penalty", + "seed", + "response_format", + "structured_outputs", + "tools", + "tool_choice", + "reasoning_effort" + ], "reasoning_policy": "disabled", "reasoning_effort": null, "catalog_reasoning": { "mandatory": false, - "supported_efforts": ["high", "none"], + "supported_efforts": [ + "high", + "none" + ], "default_effort": "high" }, - "fixed_options": {"OPENROUTER_REASONING_ENABLED": "false"}, - "absent_options": ["OPENROUTER_REASONING_EFFORT"], + "fixed_options": { + "OPENROUTER_REASONING_ENABLED": "false" + }, + "absent_options": [ + "OPENROUTER_REASONING_EFFORT" + ], "role": "Mistral European open-weight anchor" + }, + { + "id": "openrouter-deepseek-v4-flash-0731-deepseek", + "provider": "openrouter", + "model": "deepseek/deepseek-v4-flash-0731", + "canonical_slug": "deepseek/deepseek-v4-flash-20260731", + "transport": "gateway-api", + "cohort": "open-weight", + "upstream_provider": "DeepSeek", + "upstream_provider_slug": "deepseek/fp8", + "endpoint_tag": "deepseek/fp8", + "endpoint_name": "DeepSeek | deepseek/deepseek-v4-flash-20260731", + "catalog_route_status": 0, + "catalog_uptime_last_30m": 99.99164403593065, + "catalog_supported_parameters": [ + "reasoning", + "include_reasoning", + "max_tokens", + "temperature", + "top_p", + "stop", + "frequency_penalty", + "presence_penalty", + "logprobs", + "top_logprobs", + "tools", + "tool_choice", + "response_format", + "reasoning_effort" + ], + "reasoning_policy": "disabled", + "reasoning_effort": null, + "catalog_reasoning": { + "mandatory": false, + "default_enabled": true, + "supported_efforts": [ + "max", + "high", + "low" + ], + "default_effort": "high" + }, + "fixed_options": { + "OPENROUTER_REASONING_ENABLED": "false" + }, + "absent_options": [ + "OPENROUTER_REASONING_EFFORT" + ], + "role": "DeepSeek open-weight anchor on the first-party FP8 route" + }, + { + "id": "openrouter-hy3-tencent", + "provider": "openrouter", + "model": "tencent/hy3", + "canonical_slug": "tencent/hy3-20260706", + "transport": "gateway-api", + "cohort": "open-weight", + "upstream_provider": "Tencent", + "upstream_provider_slug": "tencent/fp8", + "endpoint_tag": "tencent/fp8", + "endpoint_name": "Tencent | tencent/hy3-20260706", + "catalog_route_status": 0, + "catalog_uptime_last_30m": 99.88329118848473, + "catalog_supported_parameters": [ + "reasoning", + "include_reasoning", + "temperature", + "stop", + "max_completion_tokens", + "max_tokens", + "response_format", + "tools", + "tool_choice", + "reasoning_effort" + ], + "reasoning_policy": "disabled", + "reasoning_effort": null, + "catalog_reasoning": { + "mandatory": false, + "default_enabled": true, + "supported_efforts": [ + "high", + "low", + "none" + ], + "default_effort": "high" + }, + "fixed_options": { + "OPENROUTER_REASONING_ENABLED": "false" + }, + "absent_options": [ + "OPENROUTER_REASONING_EFFORT" + ], + "role": "Tencent open-weight anchor on the first-party FP8 route" } ], "exact_route_acceptance": { "schema_version": 1, "status": "unresolved", "entries": { - "openrouter-gpt-5.6-luna-pro-openai": { + "openrouter-gpt-5.6-luna-openai": { "route_identity_sha256": null, "authenticated": false, "verified_at_utc": null, @@ -350,18 +610,52 @@ "accepted_at_utc": null, "evidence_sha256": null } + }, + "openrouter-deepseek-v4-flash-0731-deepseek": { + "route_identity_sha256": null, + "authenticated": false, + "verified_at_utc": null, + "route_evidence_sha256": null, + "privacy_acceptance": { + "status": "unresolved", + "route_identity_sha256": null, + "data_collection_policy_accepted": false, + "retention_policy_accepted": false, + "training_use_policy_accepted": false, + "zero_data_retention_policy_accepted": false, + "accepted_at_utc": null, + "evidence_sha256": null + } + }, + "openrouter-hy3-tencent": { + "route_identity_sha256": null, + "authenticated": false, + "verified_at_utc": null, + "route_evidence_sha256": null, + "privacy_acceptance": { + "status": "unresolved", + "route_identity_sha256": null, + "data_collection_policy_accepted": false, + "retention_policy_accepted": false, + "training_use_policy_accepted": false, + "zero_data_retention_policy_accepted": false, + "accepted_at_utc": null, + "evidence_sha256": null + } } } }, "required_smokes": [ - "openrouter-gpt-5.6-luna-pro-openai", + "openrouter-gpt-5.6-luna-openai", "openrouter-claude-sonnet-5-bedrock", "openrouter-gemini-3.6-flash-google-ai-studio", "openrouter-grok-4.5-xai", "openrouter-glm-5.2-novita", "openrouter-minimax-m3-minimax", "openrouter-qwen3.7-plus-alibaba", - "openrouter-mistral-medium-3.5-mistral" + "openrouter-mistral-medium-3.5-mistral", + "openrouter-deepseek-v4-flash-0731-deepseek", + "openrouter-hy3-tencent" ], "shared_fixed_options": { "OPENROUTER_PROVIDER_SORT": "price", @@ -390,10 +684,9 @@ "Supported-parameter lists do not prove behavioral acceptance; every route still requires an authenticated zero-completion-call preflight and then a separately authorized strict smoke." ], "unresolved_decisions": [ - "Generate the already selected 15-seed private panel with the audited uniform generator and commit only its salted hiding commitment plus ordered execution hash.", + "Generate the selected 16-seed private panel with the audited uniform generator and commit only its salted hiding commitment plus ordered execution hash.", "Resolve exact-route privacy, retention, training-use, and zero-data-retention requirements.", "Run an authenticated zero-completion-call exact-route preflight only after separate operator authorization.", - "Resolve the Luna Pro description-versus-structured-reasoning metadata inconsistency.", "Validate the provisional common 4,096-token cap and 3,072-token cap-pressure rule with separately authorized strict smokes; if triggered, invalidate every v3 smoke and amend the cap symmetrically before panel data." ] } diff --git a/config/sota_v3_pricing_snapshot.json b/config/sota_v3_pricing_snapshot.json index 58d09ce..296ce1e 100644 --- a/config/sota_v3_pricing_snapshot.json +++ b/config/sota_v3_pricing_snapshot.json @@ -3,7 +3,7 @@ "contract": "sota-v3", "contract_fingerprint": "a523bdfcebe47bbd", "status": "catalog-frozen-public-metadata-only", - "checked_at_utc": "2026-07-28T04:15:49Z", + "checked_at_utc": "2026-08-03T15:53:59Z", "source": "Unauthenticated HTTP GET of https://openrouter.ai/api/v1/models and each selected model's https://openrouter.ai/api/v1/models/{model_id}/endpoints response; no completion or chat endpoint was called.", "currency": "USD", "rates_are_per_token": true, @@ -25,66 +25,78 @@ "anthropic/claude-sonnet-5": { "provider_slug": "amazon-bedrock/global", "endpoint_name": "Amazon Bedrock | anthropic/claude-sonnet-5-20260630", - "prompt": 0.000002, - "completion": 0.00001 + "prompt": 2e-06, + "completion": 1e-05 + }, + "deepseek/deepseek-v4-flash-0731": { + "provider_slug": "deepseek/fp8", + "endpoint_name": "DeepSeek | deepseek/deepseek-v4-flash-20260731", + "prompt": 1.4e-07, + "completion": 2.8e-07 }, "google/gemini-3.6-flash": { "provider_slug": "google-ai-studio", "endpoint_name": "Google AI Studio | google/gemini-3.6-flash-20260721", - "prompt": 0.0000015, - "completion": 0.0000075, - "internal_reasoning": 0.0000075 + "prompt": 1.5e-06, + "completion": 7.5e-06, + "internal_reasoning": 7.5e-06 }, "minimax/minimax-m3": { "provider_slug": "minimax/fp8", "endpoint_name": "Minimax | minimax/minimax-m3-20260531", - "prompt": 0.0000003, - "completion": 0.0000012 + "prompt": 3e-07, + "completion": 1.2e-06 }, "mistralai/mistral-medium-3-5": { "provider_slug": "mistral", "endpoint_name": "Mistral | mistralai/mistral-medium-3.5-20260430", - "prompt": 0.0000015, - "completion": 0.0000075 + "prompt": 1.5e-06, + "completion": 7.5e-06 }, - "openai/gpt-5.6-luna-pro": { + "openai/gpt-5.6-luna": { "provider_slug": "openai", - "endpoint_name": "OpenAI | openai/gpt-5.6-luna-pro-20260709", - "prompt": 0.0000005, - "completion": 0.000003, + "endpoint_name": "OpenAI | openai/gpt-5.6-luna-20260709", + "prompt": 1e-07, + "completion": 6e-07, "long_context_override": { "min_prompt_tokens": 272000, - "prompt": 0.000001, - "completion": 0.0000045 + "prompt": 2e-07, + "completion": 9e-07 } }, "qwen/qwen3.7-plus": { "provider_slug": "alibaba", "endpoint_name": "Alibaba | qwen/qwen3.7-plus-20260602", - "prompt": 0.00000032, - "completion": 0.00000128, + "prompt": 3.2e-07, + "completion": 1.28e-06, "long_context_override": { "min_prompt_tokens": 256000, - "prompt": 0.00000096, - "completion": 0.00000384 + "prompt": 9.6e-07, + "completion": 3.84e-06 } }, + "tencent/hy3": { + "provider_slug": "tencent/fp8", + "endpoint_name": "Tencent | tencent/hy3-20260706", + "prompt": 1.32e-07, + "completion": 5.28e-07 + }, "x-ai/grok-4.5": { "provider_slug": "xai/zdr", "endpoint_name": "xAI | x-ai/grok-4.5-20260708", - "prompt": 0.000002, - "completion": 0.000006, + "prompt": 2e-06, + "completion": 6e-06, "long_context_override": { "min_prompt_tokens": 200000, - "prompt": 0.000004, - "completion": 0.000012 + "prompt": 4e-06, + "completion": 1.2e-05 } }, "z-ai/glm-5.2": { "provider_slug": "novita/fp8", "endpoint_name": "Novita | z-ai/glm-5.2-20260616", - "prompt": 0.0000007266, - "completion": 0.0000022836 + "prompt": 7.266e-07, + "completion": 2.2836e-06 } }, "public_metadata_limitations": [ diff --git a/config/sota_v3_publication_protocol.json b/config/sota_v3_publication_protocol.json index 27bdadf..ad32fd2 100644 --- a/config/sota_v3_publication_protocol.json +++ b/config/sota_v3_publication_protocol.json @@ -8,13 +8,19 @@ "panel_design": { "status": "frozen", "evaluated_grid": { - "seed_range": [9, 16], - "repeat_range": [1, 3], - "holm_family_size": 8, + "seed_range": [ + 9, + 16 + ], + "repeat_range": [ + 1, + 3 + ], + "holm_family_size": 10, "simulation_trials_per_allocation": 10000 }, "selection_rule": "Smallest allocation in the predeclared 9-16 seed and 1-3 repeat grid whose conservative-sensitivity familywise all-reject power Wilson 95% lower bound is at least 0.80.", - "result": "15 seeds x 1 repeat (15 episodes/model) qualifies; sensitivity power 0.8357 with Wilson 95% CI [0.828309, 0.842834] and base power 0.9461 with Wilson 95% CI [0.9415, 0.950357].", + "result": "16 seeds x 1 repeat (16 episodes/model) qualifies; sensitivity power 0.8488 with Wilson 95% CI [0.841645, 0.855688] and base power 0.9527 with Wilson 95% CI [0.948363, 0.95669].", "seed_identity_status": "pending-authorized-generation" }, "output_policy": { @@ -41,10 +47,10 @@ "multiplicity_method": "holm-bonferroni", "alpha": 0.05, "multiplicity": "Holm-Bonferroni across the final registered model family", - "holm_family_size": 8, + "holm_family_size": 10, "target_effect_score_points": -100, "target_familywise_all_reject_power": 0.8, - "power_definition": "Probability all eight Holm-adjusted model-versus-pick-trader contrasts reject when every true lift is -100 score points. The test is two-sided; the sign is a planning assumption drawn from the frozen sota-v2 evidence, not a one-sided test.", + "power_definition": "Probability all ten Holm-adjusted model-versus-pick-trader contrasts reject when every true lift is -100 score points. The test is two-sided; the sign is a planning assumption drawn from the frozen sota-v2 evidence, not a one-sided test.", "power_model": { "distribution": "Normal parametric Monte Carlo over seed-level lifts, evaluated with the production analyzer's exact sign-flip and Holm functions.", "historical_evidence_contract": "sota-v2", @@ -58,14 +64,20 @@ "sensitivity_interaction_variance_floor_as_noise_fraction": 0.1, "simulation_trials": 10000, "simulation_seed": 2026072800, - "selected_allocation": "15 seeds x 1 repeat", - "selected_base_power": 0.9461, - "selected_base_power_wilson_ci95": [0.9415, 0.950357], - "selected_sensitivity_power": 0.8357, - "selected_sensitivity_power_wilson_ci95": [0.828309, 0.842834] + "selected_allocation": "16 seeds x 1 repeat", + "selected_base_power": 0.9527, + "selected_base_power_wilson_ci95": [ + 0.948363, + 0.95669 + ], + "selected_sensitivity_power": 0.8488, + "selected_sensitivity_power_wilson_ci95": [ + 0.841645, + 0.855688 + ] }, "ranking_rule": "No model tiers or ordinal ranking; only each registered model's predeclared contrast versus pick-trader is supported.", - "remaining_blocker": "Seed identity is not frozen. The 15-seed private panel must be generated and its salted commitment hash committed under separate owner authorization before any provider call." + "remaining_blocker": "Seed identity is not frozen. The 16-seed private panel must be generated and its salted commitment hash committed under separate owner authorization before any provider call." }, "budget_policy": { "provider": "openrouter", diff --git a/docs/PUBLISH_READINESS.md b/docs/PUBLISH_READINESS.md index e9308a9..47b3443 100644 --- a/docs/PUBLISH_READINESS.md +++ b/docs/PUBLISH_READINESS.md @@ -7,7 +7,7 @@ > to preserve this first draft; the goal is to make it more accurate as the > project develops. -**Last reviewed:** 2026-07-30 +**Last reviewed:** 2026-08-03 **Current target:** Preserve the published `sota-v2` study as frozen historical evidence while pre-registering and rehearsing a finite `sota-v3` publication lane. @@ -22,14 +22,14 @@ and every eligible model trails `pick-trader`. The three P0 correctness and artifact-integrity fixes landed in #85 as an explicit `sota-v3` contract. Sunday's merged work closed the contract-economics, same-view, gap-diagnostic, site-framing, statistical-tiering, and version-dispatched CI items. The current -working tree contains an eight-model public-catalog cohort, a frozen 15-seed x +working tree contains a ten-model public-catalog cohort, a frozen 16-seed x 1-repeat statistical design, a provisional 4,096-token safety ceiling, an empty not-started smoke manifest, explicit runner dispatch, and a zero-spend synthetic rehearsal. Exact authenticated routes, privacy acceptance, model-specific -reasoning compatibility, private seed identity, and the clean-checkout -fingerprint-bound rehearsal remain unresolved. All route, spend, execution, and -publication gates remain false. There is no real v3 smoke or leaderboard -artifact, and no paid v3 smoke or panel spend is authorized. +reasoning compatibility, and private seed identity remain unresolved. All +route, spend, execution, and publication gates remain false. There is no real +v3 smoke or leaderboard artifact, and no paid v3 smoke or panel spend is +authorized. **Current weekly focus:** [#93 — v3 readiness program: consultant audit findings](https://github.com/nedcut/gm-bench/issues/93). Remaining [Issue #84](https://github.com/nedcut/gm-bench/issues/84) follow-through is @@ -569,9 +569,12 @@ These should be refined, not quietly removed: offline rehearsal. Additional realism polish is deferred by the v3 mechanics-freeze decision below. - [x] **Run `scaffold-view` under the current candidate contract fingerprint - before a paid panel.** Completed 2026-07-25 on seeds 11–18 at 5 seasons under - fingerprint `4f6ddddd6a6dd81c`; paired mean gap versus `pick-trader` is +2.8 - points. + before a paid panel.** First measured 2026-07-25 on seeds 11–18 at 5 seasons + under fingerprint `4f6ddddd6a6dd81c`, and **revalidated 2026-08-03 under the + `sota-v3` candidate fingerprint `a523bdfcebe47bbd`** at + `88ab7df191ef4d8e3ec4921a3e374e51d7fcc91c`, where every headline mean, + per-seed score, and the paired *t* reproduce exactly. Paired mean gap versus + `pick-trader` is +2.8 points. See [`docs/run_logs/scaffold-view-official-panel-2026-07-25.md`](run_logs/scaffold-view-official-panel-2026-07-25.md). The baseline remains outside `PRESETS["leaderboard"]` (2026-07-24 entry); the gap is diagnostic only and does not re-rank models. @@ -629,15 +632,26 @@ snapshot is in [`docs/run_logs/sota-v3-design-amendment-2026-07-28.md`](run_logs/sota-v3-design-amendment-2026-07-28.md), superseding [`docs/run_logs/sota-v3-statistical-design-audit-2026-07-28.md`](run_logs/sota-v3-statistical-design-audit-2026-07-28.md). - Seed identity remains unfrozen: the 15-seed private panel requires separate + Seed identity remains unfrozen: the private seed panel requires separate owner authorization before generation with the audited unbiased generator, and both execution gates test against the literal `frozen`, so provider execution stays locked. The overall preregistration status remains `provisional-pre-smoke` until route, privacy, reasoning, seed, and rehearsal - evidence is complete. The reproducible conservative pre-smoke reservation is - recorded in - `results/analysis/sota-v3-pre-smoke-cost-estimate.json`: 2,432 calls, - $86.592730 before contingency and $103.911276 at the committed 1.2x reserve. + evidence is complete. Design amendment 2 (2026-08-03, + [`docs/run_logs/sota-v3-design-amendment-2026-08-03.md`](run_logs/sota-v3-design-amendment-2026-08-03.md)) + is a pre-data cohort update: the OpenAI anchor moves from GPT-5.6 Luna Pro to + plain GPT-5.6 Luna, and DeepSeek V4 Flash 0731 plus Tencent Hy3 join as + open-weight anchors on first-party FP8 routes (Thinking Machines Inkling + Small was evaluated and recorded ineligible: no healthy route advertises + `response_format` under the lane's frozen JSON-mode and require-parameters + options). The Holm family is now ten, and the allocation reselected with the + identical frozen machinery is **16 seeds x 1 repeat (16 episodes/model)** — + base power 0.9527, sensitivity 0.8488 (Wilson 95% lower bound 0.8416); the + prior 15x1 allocation fails the lower-bound rule at family ten (0.8020, + lower bound 0.7941). The contract fingerprint is unchanged. The reproducible + conservative pre-smoke reservation is recorded in + `results/analysis/sota-v3-pre-smoke-cost-estimate.json`: 3,240 calls, + $89.845094 before contingency and $107.814113 at the committed 1.2x reserve. Runtime remains pending accepted smoke telemetry, so this does not set an operator ceiling or authorize spend. A runtime private panel can change without changing the contract fingerprint; @@ -666,19 +680,26 @@ snapshot is in panel evidence. - [x] After these changes are committed, rerun the rehearsal from a clean checkout at the exact candidate SHA and record that SHA before any spend. - **Verified unassisted 2026-08-01 at candidate SHA - `63f28e6897383fe73394c1e4354005dd404fb30b`.** A `--no-local` clone with no - `web/node_modules` runs `python3 scripts/sota_v3_rehearsal.py` to completion - with no manual preparation: status `passed`, `spend_usd` 0.0, + **Verified unassisted 2026-08-03 at candidate SHA + `3be9432e10dd0f81c9e58f89bbaae1e0c5a7465f`.** A fresh clone from the GitHub + remote with no `web/node_modules` runs `python3 scripts/sota_v3_rehearsal.py` + to completion with no manual preparation: status `passed`, `spend_usd` 0.0, `evidence_class` `synthetic-non-evidence`, all seven mutations rejected, policy selection `sota_v2` rejected / `sota_v3` accepted, generated site data byte-matching the checked-in frozen v2 dataset with the synthetic v3 row excluded, dependencies `installed` via `bun install --frozen-lockfile` (40 packages), and a successful staged build. The full suite passes in the - same clone (738 tests). No contract source was touched, so the fingerprint + same clone (741 tests). No contract source was touched, so the fingerprint remains `a523bdfcebe47bbd`, matching `config/sota_v3_lane.json`. This authorizes no spend; every lane gate remains false. + This re-verification supersedes the 2026-08-01 run at + `63f28e6897383fe73394c1e4354005dd404fb30b`, which recorded the same result + but predates the cohort-ten amendment. That amendment edits + `config/sota_v3_lane.json` and `config/sota_v3_models.json`, both of which + the rehearsal validates against, so the earlier SHA no longer evidences the + configuration a paid run would use. + The first attempt, at `a0fdec5493eaf5f702e71c519910e03f39e727f5`, **failed** — recorded here because the failure mode is the point. `_run_web_build` aborted with an unhandled `CalledProcessError` (exit 127, `vite` absent) @@ -692,6 +713,16 @@ snapshot is in `_ensure_web_dependencies` now resolves the dependency explicitly and records under `web_build.dependencies` which path ran; three regression tests cover install, reuse, and failure. + + Those three tests shipped incomplete, and the gap is worth recording because + it repeats the pattern above. All three call `_ensure_web_dependencies` + directly, so none of them touches the call site inside `_run_web_build` — + the single line that makes a clean checkout work. Stubbing that line out + left the suite fully green, meaning the fix could be deleted without any + signal. Two tests added 2026-08-03 drive `_run_web_build` itself and assert + the install precedes the build; the same sabotage now fails 2 of 11 tests. + A regression test that passes with the code removed is not coverage, and the + only way to learn which case you are in is to break the code on purpose. - [ ] Select exact routes, grant the separate zero-completion-call `route_preflight_authorized` gate, then run `python3 scripts/run_publication_matrix.py route-preflight --contract sota-v3`. @@ -699,7 +730,7 @@ snapshot is in subprocess, reserve spend, or create run state. - [ ] After explicit owner authorization, use `scripts/seed_panel_commitment.py generate --lane config/sota_v3_lane.json - --secret-file ` to draw the ordered 15-seed panel + --secret-file ` to draw the ordered 16-seed panel uniformly with `secrets.randbelow`, excluding duplicates and committed public preset seeds. Independently verify the salted hiding commitment and ordered execution hash; commit commitments only, never seed values. The legacy @@ -732,9 +763,11 @@ invalidation of v3 preregistration evidence tied to the prior fingerprint, and free diagnostic re-runs before any spend. The earlier `scaffold-view` diagnostic was measured under -`4f6ddddd6a6dd81c`; it must be rerun or explicitly revalidated under the current -fingerprint before an accepted paid smoke. This does not pre-decide the -publication-lane parameters that still must be registered: authenticated +`4f6ddddd6a6dd81c` and was revalidated 2026-08-03 under the current fingerprint +`a523bdfcebe47bbd` with every figure reproducing exactly, so this gate is +closed for the current contract; any future fingerprint change reopens it. This +does not pre-decide the publication-lane parameters that still must be +registered: authenticated route/privacy acceptance, model-specific reasoning compatibility, private seed-panel identity, exclusions, operator ceiling, and site treatment. The provisional output policy is a 4,096-token fixed safety ceiling with a 3,072 @@ -856,7 +889,7 @@ than pasting large outputs. | 2026-07-18 | Final Fable 5 launch audit | Conditions resolved pre-data | `docs/run_logs/sota-v2-final-launch-audit-2026-07-18.md` | No P0 blocker. Reconciled the stale output-policy text, strengthened reservations for repairs plus contingency, selected a $95 operator ceiling, and retained Tencent timing and per-cell spend monitoring as launch conditions. | | 2026-07-24 | P0 integrity hardening and v3 boundary | Merged | [#85](https://github.com/nedcut/gm-bench/pull/85) | Fixes non-finite actions, negotiation-window resets, and compact-artifact integrity without mutating the frozen v2 release contract. Merged as `1e5cd44`. | | 2026-07-24 | #84 P1: score decomposition, strict publication fallback, same-view reference | Merged | [#88](https://github.com/nedcut/gm-bench/pull/88) | Persists `score_components` on every episode row, makes strict failure handling the resolved-and-recorded publication default, and registers the `scaffold-view` diagnostic. All three are `sota-v3`-only; no v2 artifact is touched. The scaffold-view diagnostic panel ran 2026-07-25 (see run log); no paid v3 model panel has run. | -| 2026-07-25 | scaffold-view official panel measurement | Complete | [`docs/run_logs/scaffold-view-official-panel-2026-07-25.md`](run_logs/scaffold-view-official-panel-2026-07-25.md) / [#98](https://github.com/nedcut/gm-bench/pull/98) | Deterministic compare vs `pick-trader` on seeds 11–18 × 5 seasons under fingerprint `4f6ddddd6a6dd81c` (re-measured after PR #92 polish; scores unchanged). Paired mean gap +2.8 (six seeds tied; only seeds 17–18 diverge). Diagnostic only. | +| 2026-07-25 | scaffold-view official panel measurement | Complete | [`docs/run_logs/scaffold-view-official-panel-2026-07-25.md`](run_logs/scaffold-view-official-panel-2026-07-25.md) / [#98](https://github.com/nedcut/gm-bench/pull/98) | Deterministic compare vs `pick-trader` on seeds 11–18 × 5 seasons under fingerprint `4f6ddddd6a6dd81c` (re-measured after PR #92 polish; scores unchanged). Paired mean gap +2.8 (six seeds tied; only seeds 17–18 diverge). Revalidated 2026-08-03 under the `sota-v3` candidate fingerprint `a523bdfcebe47bbd`; every per-seed score and the paired *t* reproduce exactly. Diagnostic only. | | 2026-07-26 | Gap decomposition and panel power | Complete | [`docs/run_logs/gap-decomposition-and-panel-power-2026-07-26.md`](run_logs/gap-decomposition-and-panel-power-2026-07-26.md) | Protocol friction bounds at 0.5–9.0% of the model-vs-`pick-trader` gap; fresh-spawn/memo-only continuity costs the scripted references exactly zero (now enforced by `tests/test_reference_statelessness.py`); memo-write volume is not meaningfully associated with score (not a causal ablation). Variance decomposition: within-seed noise sd 53.4 vs seed difficulty sd 13.45, model×seed interaction indistinguishable from zero. For the published eight-model family, the Holm illustration (matching `model_tiers.py`) needs ~96 episodes/model for 0.95 power at Δ=40, not 48; rerun after the v3 family is selected. No contract source touched; no spend authorised. | | 2026-07-27 | v3 readiness reconciliation and mechanics freeze | Ready for review | [`docs/run_logs/sota-v3-preflight-2026-07-27.md`](run_logs/sota-v3-preflight-2026-07-27.md) | Confirms PR #99 CI dispatch is present; records the provisional, fail-closed v3 config/runner package and passing zero-spend rehearsal; keeps v2 as the public evidence lane; freezes mechanics; and authorizes no paid spend. A post-commit clean-checkout rerun remains required. | | 2026-07-24 | Results-first public site | Merged | [#87](https://github.com/nedcut/gm-bench/pull/87) | Reframes the public result around one unresolved model tier, the scripted-reference gap, compute, and auditability. | diff --git a/docs/run_logs/scaffold-view-official-panel-2026-07-25.md b/docs/run_logs/scaffold-view-official-panel-2026-07-25.md index b58c99e..909bfb8 100644 --- a/docs/run_logs/scaffold-view-official-panel-2026-07-25.md +++ b/docs/run_logs/scaffold-view-official-panel-2026-07-25.md @@ -76,3 +76,29 @@ polish fixes that moved the fingerprint from `0a5f0434dca31ac5` to `feat/contract-economics` tip and the compare was re-run. Headline means, per-seed scores, and paired *t* are unchanged on the official panel; only the fingerprint and economics commit reference above were updated. + +## Revalidation under the `sota-v3` candidate fingerprint (2026-08-03) + +`docs/PUBLISH_READINESS.md` requires this diagnostic to be rerun or explicitly +revalidated under the fingerprint a paid smoke would actually run on. The +figures above were measured under `4f6ddddd6a6dd81c`; the `sota-v3` candidate +contract is `a523bdfcebe47bbd`. The compare was therefore re-run unchanged: + +| Field | Value | +| --- | --- | +| Contract fingerprint | `a523bdfcebe47bbd` | +| Branch / commit | `amend/sota-v3-cohort-10` at `88ab7df191ef4d8e3ec4921a3e374e51d7fcc91c` | +| Command | `python3 -m gm_bench compare --agents scaffold-view pick-trader --seeds 11 12 13 14 15 16 17 18 --seasons 5 --no-log` | +| Spend | $0.00 — both agents are scripted and CPU-only | + +**Every number above reproduces exactly.** Headline means 270.675 and 267.875, +paired mean difference +2.800, paired *t* 0.249, and all eight per-seed scores +match the `4f6ddddd6a6dd81c` measurement to the recorded precision, including +the two divergent seeds (17: +69.935, 18: −47.533) and the six exact ties. +Neither agent reads a field the intervening contract changes touched, so the +observation-asymmetry bound carries forward unaltered. + +This revalidates the bound; it does not widen the claim. The +2.8 gap remains +diagnostic, remains driven entirely by seeds 17 and 18, and still may not be +used to re-rank any model. `scaffold-view` stays outside +`PRESETS["leaderboard"]`. diff --git a/docs/run_logs/sota-v3-design-amendment-2026-08-03.md b/docs/run_logs/sota-v3-design-amendment-2026-08-03.md new file mode 100644 index 0000000..91a3e3c --- /dev/null +++ b/docs/run_logs/sota-v3-design-amendment-2026-08-03.md @@ -0,0 +1,134 @@ +# `sota-v3` design amendment 2 — ten-model cohort, 2026-08-03 + +This is a no-spend, pre-data amendment. No provider or completion endpoint was +called, no private seeds were generated, and every execution, spend, and +publication authorization remains false. It supersedes the cohort and +allocation frozen in `sota-v3-design-amendment-2026-07-28.md`; everything not +restated here carries forward unchanged from that record. + +## Why the cohort changed + +This is an owner-directed cohort update made while `evidence_state` is still +`pre-data-no-v3-model-smokes-or-panel-results`, so no observed v3 score could +have influenced it. + +**The OpenAI anchor moves from GPT-5.6 Luna Pro to plain GPT-5.6 Luna.** The +2026-07-28 registry preserved a substitution to the Pro variant made when the +plain Luna route was unhealthy. At the 2026-08-03 catalog snapshot the plain +route is healthy (status 0, 99.3% uptime over 30 minutes), and the Pro variant +carried an unresolved inconsistency between its public description +(`reasoning.mode=pro`) and the structured catalog (reasoning optional, +supports `none`). Restoring the plain route removes that ambiguity from the +registry; the corresponding `unresolved_decisions` item is dropped as +resolved-by-substitution. + +**Two open-weight anchors join the family.** DeepSeek V4 Flash 0731 and +Tencent Hy3, both on first-party FP8 routes. Both routes are healthy at the +snapshot, neither has mandatory reasoning, and both are pinned under the +cohort-wide disabled-reasoning policy. The cohort balance moves from 4/4 to 4 +frontier-proprietary / 6 open-weight. + +**Thinking Machines Inkling Small was evaluated for the tenth slot and is +ineligible at this snapshot.** The frozen lane runs every model with +`OPENROUTER_JSON_MODE=true` (the adapter sends +`response_format={"type":"json_object"}`) and +`OPENROUTER_REQUIRE_PARAMETERS=true`, so OpenRouter only routes to endpoints +that advertise `response_format`. Neither of inkling-small's catalog routes +advertises it, and the full-size inkling's only advertising route (DeepInfra) +is degraded at the snapshot while the registry requires a healthy pinned +route. The ineligibility is recorded here so a later route change can be +revisited only through another pre-data amendment. Moonshot Kimi K3 was also +evaluated and passed the route requirements but was declined on cost: at its +pinned first-party rate it would have been the most expensive model in the +panel. + +## Why the allocation changed + +A family of ten tightens the Holm first-step threshold from `0.05/8 = +0.00625` to `0.05/10 = 0.005`, and the familywise all-reject event over ten +contrasts is strictly harder than over eight under the frozen +compound-symmetry covariance. The allocation was reselected with the identical +frozen machinery — same historical variance components, planning effect -100, +sensitivity multipliers, 9-16 seed by 1-3 repeat grid, 10,000 trials, +simulation seed 2026072800 — changing only `--family-size` from 8 to 10. + +The previously selected 15 seeds x 1 repeat no longer qualifies at family +ten: its conservative-sensitivity familywise power is 0.8020 with Wilson 95% +CI [0.794074, 0.809694], and the predeclared rule requires the *lower bound* +to clear 0.80. The smallest qualifying allocation is: + +| Item | Family of eight (superseded) | Family of ten (this amendment) | +|---|---|---| +| Holm first-step threshold | 0.00625 | 0.005 | +| Selected allocation | 15 seeds x 1 repeat | **16 seeds x 1 repeat** | +| Episodes per model | 15 | **16** | +| Base power (Wilson 95%) | 0.9461 [0.9415, 0.950357] | 0.9527 [0.948363, 0.95669] | +| Sensitivity power (Wilson 95%) | 0.8357 [0.828309, 0.842834] | 0.8488 [0.841645, 0.855688] | +| Minimum exact two-sided p | 6.103515625e-05 | 3.0517578125e-05 | +| Total panel episodes | 120 | 160 | + +At 16 seeds the exact sign-flip resolution floor `2/2^16` sits far below the +0.005 Holm first step, so exact feasibility is preserved. + +The pending private seed panel is now 16 seeds. Its identity remains +unfrozen: `seed_panel.name` and `seed_panel.sha256` are null until the panel +is generated under separate owner authorization and only its salted hiding +commitment plus ordered execution hash are committed. + +## Cost consequence + +The regenerated pre-smoke reservation +(`results/analysis/sota-v3-pre-smoke-cost-estimate.json`) for the 10-model, +16-seed panel plus required smokes is $89.85 unrounded, $107.81 with the 1.2x +contingency multiplier (previously $86.59 / $103.91). Both added models price +below the cohort median, and plain Luna is cheaper than Luna Pro at the pinned +base rates, so the cheaper substitution partly offsets the extra seed and the +two added smoke-plus-panel lanes. This remains a planning reservation, not an +authorization; the operator ceiling is still null and `spend_authorized` is +still false. + +## Unchanged + +- Directional primary claim: every registered model-plus-compact-scaffold + system trails deterministic `pick-trader` on seed-paired mean lift. +- alpha 0.05, Holm-Bonferroni across the fixed registered family, exact + two-sided enumeration sign-flip test, seed as the unit of inference. +- Planning effect -100 score points; 0.80 conservative-sensitivity + familywise all-reject power target with the Wilson lower-bound rule. +- Contract fingerprint `a523bdfcebe47bbd`; no `_CONTRACT_SOURCES` file was + touched. +- The provisional 4,096-token output ceiling, 3,072 cap-pressure trigger, + and symmetric amendment rule. +- No model-to-model tiers or ordinal ranking. +- Every execution, spend, and publication authorization remains false, and + route preflight, seed generation, smokes, panel, and publication each still + require their own separate owner decisions. + +## Reproduction + +```bash +python3 scripts/panel_power.py \ + --exact-reference-family \ + --family-size 10 \ + --delta -100 \ + --target-power 0.80 \ + --min-seeds 9 \ + --max-seeds 16 \ + --max-repeats 3 \ + --trials 10000 \ + --json + +python3 scripts/estimate_publication_cost.py \ + --models-config config/sota_v3_models.json \ + --lane-config config/sota_v3_lane.json \ + --pricing config/sota_v3_pricing_snapshot.json \ + --output results/analysis/sota-v3-pre-smoke-cost-estimate.json + +python3 -m pytest -q \ + tests/test_panel_power.py \ + tests/test_publication_cost.py \ + tests/test_publication_analysis.py \ + tests/test_seed_panel_commitment.py \ + tests/test_sota_v3_preregistration.py \ + tests/test_sota_v3_route_catalog.py +``` diff --git a/results/analysis/sota-v3-pre-smoke-cost-estimate.json b/results/analysis/sota-v3-pre-smoke-cost-estimate.json index a8ac55f..fe7c1a9 100644 --- a/results/analysis/sota-v3-pre-smoke-cost-estimate.json +++ b/results/analysis/sota-v3-pre-smoke-cost-estimate.json @@ -8,7 +8,7 @@ "panel_preset": "leaderboard", "panel_repeats": 1, "panel_seasons": 5, - "panel_seed_count": 15, + "panel_seed_count": 16, "phase_count": 4, "rates_are_per_token": true, "serial_workers": 1, @@ -17,33 +17,33 @@ "smoke_seed_count": 1 }, "calls": { - "model_count": 8, - "panel_calls": 2400, - "panel_decisions_per_model": 300, - "smoke_calls": 32, + "model_count": 10, + "panel_calls": 3200, + "panel_decisions_per_model": 320, + "smoke_calls": 40, "smoke_decisions_per_run": 4, - "smoke_runs": 8, - "total_calls": 2432 + "smoke_runs": 10, + "total_calls": 3240 }, "costs_usd": { - "panel": 85.45335168, - "smoke": 1.1393780224, - "total_unrounded": 86.5927297024, - "total_with_1_2x_contingency": 103.91127564288 + "panel": 88.735895552, + "smoke": 1.1091986944, + "total_unrounded": 89.8450942464, + "total_with_1_2x_contingency": 107.81411309568 }, "models": [ { - "applied_completion_rate_usd": 3e-06, + "applied_completion_rate_usd": 6e-07, "applied_internal_reasoning_rate_usd": 0.0, - "applied_prompt_rate_usd": 5e-07, - "cost_per_decision_usd": 0.016288, - "experiment_id": "openrouter-gpt-5.6-luna-pro-openai", + "applied_prompt_rate_usd": 1e-07, + "cost_per_decision_usd": 0.0032576, + "experiment_id": "openrouter-gpt-5.6-luna-openai", "internal_reasoning_tokens_per_decision": 0, - "model": "openai/gpt-5.6-luna-pro", - "panel_calls": 300, - "panel_cost_usd": 4.8864, + "model": "openai/gpt-5.6-luna", + "panel_calls": 320, + "panel_cost_usd": 1.042432, "smoke_calls": 4, - "smoke_cost_usd": 0.065152 + "smoke_cost_usd": 0.0130304 }, { "applied_completion_rate_usd": 1e-05, @@ -53,8 +53,8 @@ "experiment_id": "openrouter-claude-sonnet-5-bedrock", "internal_reasoning_tokens_per_decision": 0, "model": "anthropic/claude-sonnet-5", - "panel_calls": 300, - "panel_cost_usd": 17.088, + "panel_calls": 320, + "panel_cost_usd": 18.2272, "smoke_calls": 4, "smoke_cost_usd": 0.22784 }, @@ -67,8 +67,8 @@ "internal_reasoning_billing_basis": "internal_reasoning", "internal_reasoning_tokens_per_decision": 4096, "model": "google/gemini-3.6-flash", - "panel_calls": 300, - "panel_cost_usd": 22.032, + "panel_calls": 320, + "panel_cost_usd": 23.5008, "smoke_calls": 4, "smoke_cost_usd": 0.29376 }, @@ -81,8 +81,8 @@ "internal_reasoning_billing_basis": "completion", "internal_reasoning_tokens_per_decision": 4096, "model": "x-ai/grok-4.5", - "panel_calls": 300, - "panel_cost_usd": 19.5456, + "panel_calls": 320, + "panel_cost_usd": 20.84864, "smoke_calls": 4, "smoke_cost_usd": 0.260608 }, @@ -94,8 +94,8 @@ "experiment_id": "openrouter-glm-5.2-novita", "internal_reasoning_tokens_per_decision": 0, "model": "z-ai/glm-5.2", - "panel_calls": 300, - "panel_cost_usd": 4.54992768, + "panel_calls": 320, + "panel_cost_usd": 4.853256192, "smoke_calls": 4, "smoke_cost_usd": 0.0606657024 }, @@ -107,8 +107,8 @@ "experiment_id": "openrouter-minimax-m3-minimax", "internal_reasoning_tokens_per_decision": 0, "model": "minimax/minimax-m3", - "panel_calls": 300, - "panel_cost_usd": 2.19456, + "panel_calls": 320, + "panel_cost_usd": 2.340864, "smoke_calls": 4, "smoke_cost_usd": 0.0292608 }, @@ -120,8 +120,8 @@ "experiment_id": "openrouter-qwen3.7-plus-alibaba", "internal_reasoning_tokens_per_decision": 0, "model": "qwen/qwen3.7-plus", - "panel_calls": 300, - "panel_cost_usd": 2.340864, + "panel_calls": 320, + "panel_cost_usd": 2.4969216, "smoke_calls": 4, "smoke_cost_usd": 0.03121152 }, @@ -133,13 +133,39 @@ "experiment_id": "openrouter-mistral-medium-3.5-mistral", "internal_reasoning_tokens_per_decision": 0, "model": "mistralai/mistral-medium-3-5", - "panel_calls": 300, - "panel_cost_usd": 12.816, + "panel_calls": 320, + "panel_cost_usd": 13.6704, "smoke_calls": 4, "smoke_cost_usd": 0.17088 + }, + { + "applied_completion_rate_usd": 2.8e-07, + "applied_internal_reasoning_rate_usd": 0.0, + "applied_prompt_rate_usd": 1.4e-07, + "cost_per_decision_usd": 0.00226688, + "experiment_id": "openrouter-deepseek-v4-flash-0731-deepseek", + "internal_reasoning_tokens_per_decision": 0, + "model": "deepseek/deepseek-v4-flash-0731", + "panel_calls": 320, + "panel_cost_usd": 0.7254016, + "smoke_calls": 4, + "smoke_cost_usd": 0.00906752 + }, + { + "applied_completion_rate_usd": 5.28e-07, + "applied_internal_reasoning_rate_usd": 0.0, + "applied_prompt_rate_usd": 1.32e-07, + "cost_per_decision_usd": 0.003218688, + "experiment_id": "openrouter-hy3-tencent", + "internal_reasoning_tokens_per_decision": 0, + "model": "tencent/hy3", + "panel_calls": 320, + "panel_cost_usd": 1.02998016, + "smoke_calls": 4, + "smoke_cost_usd": 0.012874752 } ], - "pricing_checked_at_utc": "2026-07-28T04:15:49Z", + "pricing_checked_at_utc": "2026-08-03T15:53:59Z", "runtime": { "note": "Regenerate this artifact from accepted smoke telemetry before approving the full panel; latency is reported only for models with committed observations.", "observation_source": null, @@ -150,6 +176,6 @@ "schema_version": 2, "supersedes": { "artifact": "retired 12-cell output-budget sweep estimate", - "description": "Replaces the four-cap, three-model sweep estimate with the registered 8-model fixed 4,096-token panel and its required smoke gate." + "description": "Replaces the four-cap, three-model sweep estimate with the registered 10-model fixed 4,096-token panel and its required smoke gate." } } diff --git a/tests/test_publication_cost.py b/tests/test_publication_cost.py index 02b40e2..c8c97dd 100644 --- a/tests/test_publication_cost.py +++ b/tests/test_publication_cost.py @@ -41,13 +41,13 @@ def test_fixed_panel_and_smoke_call_counts() -> None: def test_v3_cost_plan_uses_registered_private_seed_count() -> None: result = estimate(*_v3_inputs()) - assert result["assumptions"]["panel_seed_count"] == 15 + assert result["assumptions"]["panel_seed_count"] == 16 assert result["assumptions"]["panel_repeats"] == 1 - assert result["calls"]["panel_decisions_per_model"] == 300 - assert result["calls"]["panel_calls"] == 2_400 - assert result["calls"]["total_calls"] == 2_432 - assert result["costs_usd"]["total_unrounded"] == pytest.approx(86.5927297024) - assert result["costs_usd"]["total_with_1_2x_contingency"] == pytest.approx(103.91127564288) + assert result["calls"]["panel_decisions_per_model"] == 320 + assert result["calls"]["panel_calls"] == 3_200 + assert result["calls"]["total_calls"] == 3_240 + assert result["costs_usd"]["total_unrounded"] == pytest.approx(89.8450942464) + assert result["costs_usd"]["total_with_1_2x_contingency"] == pytest.approx(107.81411309568) grok = next(row for row in result["models"] if row["model"] == "x-ai/grok-4.5") assert grok["internal_reasoning_tokens_per_decision"] == 4096 assert grok["applied_internal_reasoning_rate_usd"] == pytest.approx(grok["applied_completion_rate_usd"]) @@ -123,6 +123,22 @@ def test_model_specific_reasoning_and_long_context_rates_are_applied() -> None: assert row["internal_reasoning_billing_basis"] == "internal_reasoning" +def test_v3_luna_uses_the_pinned_long_context_price_tier() -> None: + models, lane, pricing = _v3_inputs() + models = copy.deepcopy(models) + pricing = copy.deepcopy(pricing) + model_name = "openai/gpt-5.6-luna" + models["models"] = [model for model in models["models"] if model["model"] == model_name] + pricing["planning_assumptions"]["input_tokens_per_decision"] = 272_000 + + result = estimate(models, lane, pricing) + row = result["models"][0] + + assert row["model"] == model_name + assert row["applied_prompt_rate_usd"] == pytest.approx(2e-7) + assert row["applied_completion_rate_usd"] == pytest.approx(9e-7) + + def test_internal_reasoning_price_requires_an_explicit_token_assumption() -> None: models, lane, pricing = _committed_inputs() pricing = copy.deepcopy(pricing) diff --git a/tests/test_sota_v3_preregistration.py b/tests/test_sota_v3_preregistration.py index 63ee1be..fe2cc6c 100644 --- a/tests/test_sota_v3_preregistration.py +++ b/tests/test_sota_v3_preregistration.py @@ -53,9 +53,9 @@ def test_v3_lane_pins_current_contract_and_freezes_a_powered_allocation() -> Non assert candidate["claim_direction"] == "trails-reference" assert candidate["target_effect_score_points"] == -100 selected = candidate["selected_allocation"] - assert selected["seed_count"] == 15 + assert selected["seed_count"] == 16 assert selected["repeats"] == 1 - assert selected["episodes_per_model"] == selected["seed_count"] * selected["repeats"] == 15 + assert selected["episodes_per_model"] == selected["seed_count"] * selected["repeats"] == 16 feasibility = exact_sign_flip_feasibility( selected["seed_count"], candidate["holm_family_size"], @@ -69,7 +69,7 @@ def test_v3_lane_pins_current_contract_and_freezes_a_powered_allocation() -> Non assert lane["seed_panel"] == { "status": "pending-authorized-generation", "name": None, - "count": 15, + "count": 16, "sha256": None, } assert lane["seed_panel"]["count"] == selected["seed_count"] @@ -96,7 +96,7 @@ def test_v3_registry_is_truthfully_provisional_and_contains_no_unverified_routes assert registry["selection_frozen_at_utc"] is None assert registry["catalog_snapshot_status"] == "frozen-public-metadata-only" assert registry["catalog_checked_at_utc"] - assert len(registry["models"]) == len(registry["required_smokes"]) == 8 + assert len(registry["models"]) == len(registry["required_smokes"]) == 10 assert set(registry["required_smokes"]) == {model["id"] for model in registry["models"]} assert registry["repeats"] == lane["repeats"] == 1 assert registry["output_token_cap"] == lane["output_token_cap"] == 4096 @@ -126,12 +126,12 @@ def test_v3_protocol_and_pricing_are_separate_and_fail_closed() -> None: assert protocol["statistical_analysis_plan"]["reference_agent"] == lane["reference_agent"] == "pick-trader" assert protocol["statistical_analysis_plan"]["multiplicity_method"] == "holm-bonferroni" assert protocol["statistical_analysis_plan"]["alpha"] == 0.05 - assert protocol["statistical_analysis_plan"]["holm_family_size"] == 8 + assert protocol["statistical_analysis_plan"]["holm_family_size"] == 10 assert protocol["statistical_analysis_plan"]["power_model"]["historical_shared_seed_variance"] == pytest.approx( 3770.478399 ) - assert protocol["statistical_analysis_plan"]["power_model"]["selected_sensitivity_power"] == 0.8357 - assert protocol["statistical_analysis_plan"]["power_model"]["selected_allocation"] == "15 seeds x 1 repeat" + assert protocol["statistical_analysis_plan"]["power_model"]["selected_sensitivity_power"] == 0.8488 + assert protocol["statistical_analysis_plan"]["power_model"]["selected_allocation"] == "16 seeds x 1 repeat" assert protocol["statistical_analysis_plan"]["claim_direction"] == "trails-reference" assert protocol["statistical_analysis_plan"]["target_effect_score_points"] == -100 # The lane and the protocol carry separate copies of the design; they must agree. @@ -143,7 +143,7 @@ def test_v3_protocol_and_pricing_are_separate_and_fail_closed() -> None: assert protocol["budget_policy"]["spend_authorized"] is False assert pricing["status"] == "catalog-frozen-public-metadata-only" assert pricing["checked_at_utc"] - assert len(pricing["models"]) == 8 + assert len(pricing["models"]) == 10 assert pricing["spend_authorized"] is False diff --git a/tests/test_sota_v3_rehearsal.py b/tests/test_sota_v3_rehearsal.py index ad0312f..72b0d78 100644 --- a/tests/test_sota_v3_rehearsal.py +++ b/tests/test_sota_v3_rehearsal.py @@ -471,3 +471,58 @@ def fake_run(cmd, **kwargs): with pytest.raises(RuntimeError, match="lockfile out of date"): rehearsal_mod._ensure_web_dependencies(staging, "bun") + + +def _record_bun_invocations(monkeypatch: pytest.MonkeyPatch) -> list[tuple[list[str], Path | None]]: + """Stub out `bun` discovery and execution, returning recorded `(argv, cwd)` pairs. + + The working directory is recorded because argv alone underspecifies the + invocation: an install or build aimed at the real `web/` tree instead of the + staging copy would run the same command line while defeating the point of + staging, and the rehearsal's isolation guarantee with it. + """ + calls: list[tuple[list[str], Path | None]] = [] + + def fake_run(cmd, **kwargs): + cwd = kwargs.get("cwd") + calls.append((list(cmd), Path(cwd) if cwd is not None else None)) + return subprocess.CompletedProcess(cmd, 0, stdout="✓ built in 87ms", stderr="40 packages installed") + + monkeypatch.setattr(rehearsal_mod.shutil, "which", lambda _name: "bun") + monkeypatch.setattr(rehearsal_mod.subprocess, "run", fake_run) + return calls + + +def test_web_build_installs_dependencies_before_building(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None: + """`_run_web_build` must resolve dependencies itself, not assume them. + + The three tests above call `_ensure_web_dependencies` directly, so they all + still pass when the call site inside `_run_web_build` is deleted -- which is + the only line that makes a clean checkout work. This test fails closed on + that regression by asserting the install actually precedes the build. + """ + staging = tmp_path / "staging" + (staging / "web").mkdir(parents=True) + calls = _record_bun_invocations(monkeypatch) + + result = rehearsal_mod._run_web_build(staging) + + assert calls == [ + (["bun", "install", "--frozen-lockfile"], staging / "web"), + (["bun", "run", "build"], staging / "web"), + ] + assert result["dependencies"]["status"] == "installed" + assert result["status"] == "passed" + + +def test_web_build_skips_the_install_when_dependencies_are_present( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + staging = tmp_path / "staging" + (staging / "web" / "node_modules").mkdir(parents=True) + calls = _record_bun_invocations(monkeypatch) + + result = rehearsal_mod._run_web_build(staging) + + assert calls == [(["bun", "run", "build"], staging / "web")] + assert result["dependencies"] == {"status": "reused", "source": "staged-directory"} diff --git a/tests/test_sota_v3_route_catalog.py b/tests/test_sota_v3_route_catalog.py index 725c742..06d5b8c 100644 --- a/tests/test_sota_v3_route_catalog.py +++ b/tests/test_sota_v3_route_catalog.py @@ -21,16 +21,21 @@ def test_v3_catalog_freezes_exact_balanced_cohort_without_unlocking_execution() registry = _read("sota_v3_models.json") models = registry["models"] - assert len(models) == 8 - assert len({model["id"] for model in models}) == 8 - assert len({(model["model"], model["endpoint_tag"]) for model in models}) == 8 + assert len(models) == 10 + assert len({model["id"] for model in models}) == 10 + assert len({(model["model"], model["endpoint_tag"]) for model in models}) == 10 assert {model["cohort"] for model in models} == {"frontier-proprietary", "open-weight"} assert sum(model["cohort"] == "frontier-proprietary" for model in models) == 4 - assert sum(model["cohort"] == "open-weight" for model in models) == 4 + assert sum(model["cohort"] == "open-weight" for model in models) == 6 identities = {model["model"] for model in models} - assert "openai/gpt-5.6-luna-pro" in identities - assert "openai/gpt-5.6-luna" not in identities + assert "openai/gpt-5.6-luna" in identities + assert "openai/gpt-5.6-luna-pro" not in identities + assert "deepseek/deepseek-v4-flash-0731" in identities + assert "tencent/hy3" in identities + # Evaluated for the tenth slot, ineligible at the frozen snapshot: no + # healthy route advertises response_format under require-parameters. + assert "thinkingmachines/inkling-small" not in identities assert "google/gemini-3.6-flash" in identities assert "google/gemini-3.5-flash" not in identities assert "mistralai/mistral-medium-3-5" in identities