{
  "schema_version": 1,
  "evidence_as_of": "2026-09-29T14:50:23.292673+00:00",
  "title": "Cloud and local component evidence",
  "scope": "Phipps, LiveBench and BFCL profiles and completeness are stated per row",
  "note": "Component evidence, not official Full or an overall ranking. Partial coverage and gates remain explicit; local capped profiles are not matched native-maximum comparisons.",
  "models": [
    {
      "id": "gemini-38-flash-native-max-rapid-20260911",
      "name": "Gemini 3.8 Flash",
      "model_id": "gemini-3.8-flash",
      "provider": "Google direct API",
      "reasoning": "Native thinkingLevel=high",
      "output_limit_tokens": 65536,
      "eval_date": "2026-09-11",
      "rankable": false,
      "promotion": false,
      "overall_score": null,
      "receipt_url": "/data/external/gemini-38-flash-native-max-rapid-20260911-qualified-components.json",
      "phipps": {
        "measured": 36,
        "required": 36,
        "visible_finals": 35,
        "score": null,
        "gate_count": 4,
        "judge_false_positive_count": 0,
        "hard_gate_failure_ids": [
          "interaction_direct_recommendation",
          "interaction_answer_first",
          "instruction_exact_json",
          "knowledge_current_hours"
        ],
        "note": "No valid composite: one task exhausted its three tool rounds without a visible final. Four recorded hard failures; deployment gates not cleared."
      },
      "livebench": {
        "answered": 90,
        "required": 90,
        "judgments_resolved": 89,
        "score": null,
        "bounds": [
          92.79495271284206,
          93.90606382395318
        ],
        "unresolved_ids": [
          "4b65576e7570cec86d6abb988889fb71304f60a49ae08de443410b0aaa44f55b"
        ],
        "note": "No final LiveBench score. One AMPS_Hard derivative judgment requires the unavailable official fallback. Its error placeholder is not a model failure. Bounds are category-balanced mathematical bounds, NOT a final score."
      },
      "bfcl": {
        "correct": 24,
        "required": 29,
        "score": 82.75862068965517,
        "official_full_comparable": false,
        "note": "Complete BFCL V4 Rapid: 29 scored trajectories, official scorer. This selected Rapid cohort is not official Full."
      },
      "performance": {
        "p50_s": 11.525658249971457,
        "p95_s": 79.90756287507247,
        "output_tokens_per_second": 320.4043444864751,
        "note": "LiveBench end-to-end requests. Output rate includes thinking, prefill/network overhead and differing tokenizers; NOT native decode speed or TTFT."
      },
      "cost": {
        "generation_usd": 4.5614625,
        "note": "Usage-priced estimate for all three benchmark components; not an invoice. Excludes admission calls and subscription judging."
      },
      "provenance": {
        "audit_sha256": "abc39a0c4ccc5af419f95f9f7b17b0515471856cca49b8dc6a3a9c9d5f7986a7",
        "closure_sha256": "6896e8ad30426af7e6169915b7f69f7082ec3fe2f4ec3e6c970786cf6f8e338a",
        "judge_diagnostic_sha256": null
      }
    },
    {
      "id": "opus-5-native-max-rapid-20260911",
      "name": "Claude Opus 5",
      "model_id": "claude-opus-5",
      "provider": "Anthropic direct API",
      "reasoning": "Adaptive thinking; effort=max",
      "output_limit_tokens": 128000,
      "eval_date": "2026-09-11",
      "rankable": false,
      "promotion": false,
      "overall_score": null,
      "receipt_url": "/data/external/opus-5-native-max-rapid-20260911-qualified-components.json",
      "phipps": {
        "measured": 36,
        "required": 36,
        "visible_finals": 36,
        "score": 4.03,
        "gate_count": 8,
        "judge_false_positive_count": 1,
        "hard_gate_failure_ids": [
          "interaction_direct_recommendation",
          "interaction_answer_first",
          "tool_memory_project_recall",
          "tool_web_conflict_synthesis",
          "memory_long_context_distractor",
          "memory_durable_decision_tool",
          "knowledge_current_hours",
          "coding_safe_command_help"
        ],
        "note": "Frozen canonical score retained. Eight recorded gates include one confirmed judge false positive: the rejected command worked with native macOS tools. Score unchanged; deployment gates not cleared."
      },
      "livebench": {
        "answered": 90,
        "required": 90,
        "judgments_resolved": 89,
        "score": null,
        "bounds": [
          89.644573743134,
          90.75568485424512
        ],
        "unresolved_ids": [
          "4b65576e7570cec86d6abb988889fb71304f60a49ae08de443410b0aaa44f55b"
        ],
        "note": "No final LiveBench score. One AMPS_Hard derivative judgment requires the unavailable official fallback. Its error placeholder is not a model failure. Bounds are category-balanced mathematical bounds, NOT a final score."
      },
      "bfcl": {
        "correct": 24,
        "required": 29,
        "score": 82.75862068965517,
        "official_full_comparable": false,
        "note": "Complete BFCL V4 Rapid: 29 scored trajectories, official scorer. This selected Rapid cohort is not official Full."
      },
      "performance": {
        "p50_s": 12.70872745802626,
        "p95_s": 71.58417329192162,
        "output_tokens_per_second": 87.9276771631221,
        "note": "LiveBench end-to-end requests. Output rate includes thinking, prefill/network overhead and differing tokenizers; NOT native decode speed or TTFT."
      },
      "cost": {
        "generation_usd": 9.19693,
        "note": "Usage-priced estimate for all three benchmark components; not an invoice. Excludes admission calls and subscription judging."
      },
      "provenance": {
        "audit_sha256": "a077c568e682d6fe9f01a164ff157f87e546fe9dc5da03f1f5f4fad2fd7aa4f8",
        "closure_sha256": "6896e8ad30426af7e6169915b7f69f7082ec3fe2f4ec3e6c970786cf6f8e338a",
        "judge_diagnostic_sha256": "65b20ec9fa949bb335d925d6c67c7a20cd59f24df708dcc8dc0b15250d30a4c3"
      }
    },
    {
      "id": "capstan-glm-cloud-rapid-20260923",
      "campaign_id": "capstan-cloud-rapid-20260923",
      "campaign_status": "partial",
      "status": "partial",
      "name": "GLM-5.3 Flash",
      "model_id": "z-ai/glm-5.3-flash",
      "provider": "Z.AI via OpenRouter FP8",
      "reasoning": "Native max; visible-final-only grading",
      "output_limit_tokens": 131072,
      "native_max_caveat": "Native maximum reasoning and output; 600-second request watchdog. Model-specific native profiles are not matched token-budget or local-runtime comparisons. Official Z.AI FP8 route pinned; no provider fallback. Direct account lacked prepaid balance.",
      "eval_date": "2026-09-23",
      "rankable": false,
      "promotion": false,
      "overall_score": null,
      "receipt_url": "/data/external/capstan-glm-cloud-rapid-20260923-rapid-evidence.json",
      "phipps": {
        "status": "complete",
        "measured": 36,
        "required": 36,
        "visible_finals": 36,
        "score": 4.27,
        "gate_count": 6,
        "judge_false_positive_count": 0,
        "hard_gate_failure_ids": [
          "instruction_exact_json",
          "interaction_direct_recommendation",
          "knowledge_current_hours",
          "memory_current_turn_over_store",
          "memory_long_context_distractor",
          "tool_web_service_snapshot"
        ],
        "note": null
      },
      "livebench": {
        "status": "partial",
        "answered": 40,
        "required": 90,
        "judgments_resolved": 0,
        "score": null,
        "note": "Stopped at the preset component wall limit. Retained answers are incomplete coverage, not a full-suite score."
      },
      "bfcl": {
        "status": "complete",
        "correct": 20,
        "required": 29,
        "score": 68.97,
        "official_full_comparable": false,
        "note": null
      },
      "performance": {
        "p50_s": 7.734506832901388,
        "p95_s": 100.39873629203066,
        "output_tokens_per_second": 32.46199650886086,
        "note": "Latency includes failed attempts; token rate uses usage-verified requests, includes reasoning, and is not native decode. Tokenizers and providers differ. Maximum request latency: 594.342801624909."
      },
      "cost": {
        "generation_usd": 0.09202372,
        "note": "OpenRouter provider-reported generation charge; not an invoice. Held campaign liabilities for this model: $0.066809. Admissions are separate from benchmark generation."
      },
      "provenance": {
        "campaign_summary_sha256": "585934a33d84cfe41ad72fd1e5ad537e79e85540aa3ef0ef924e4a94a18859d3",
        "evidence_hashes": {
          "policy": "cb50a3c6f9d5c70bdee5c84ed2822ad78bc850415fa7d32ab9f527c1331b5c0d",
          "ledger": "096dc364ebdd48df4d90609562b930f835c535863f1de0760a41d98b6479fa48",
          "execution": "5861820c30a6825024a93b0d428c13dfc5df1d00f84c36df620da61ae7f7d814",
          "phipps": "ec7e08b12b497b9e9cc7b898ffb34ff8aecc3b02c947929a109cf2c5c8cd974c",
          "bfcl": "8dcbd919b0eba6c52236037ec4bc412368176d6d442b41305ca9f5b3faca48a0",
          "phipps_rows": "b01eccd65f20c679c93f395153a44edb88e09f428b810c9c6ed1cc4d0bd07087"
        }
      }
    },
    {
      "id": "capstan-qwen-cloud-rapid-20260923",
      "campaign_id": "capstan-cloud-rapid-20260923",
      "campaign_status": "partial",
      "status": "partial",
      "name": "Qwen3.8 Flash",
      "model_id": "qwen3.8-flash",
      "provider": "Alibaba direct US",
      "reasoning": "Native xhigh; visible-final-only grading",
      "output_limit_tokens": 131072,
      "native_max_caveat": "Native maximum reasoning and output; 600-second request watchdog. Model-specific native profiles are not matched token-budget or local-runtime comparisons. Hosted Qwen3.8 Flash is a derivative, not verified identical to local Qwen3.8-Flash-Next weights.",
      "eval_date": "2026-09-23",
      "rankable": false,
      "promotion": false,
      "overall_score": null,
      "receipt_url": "/data/external/capstan-qwen-cloud-rapid-20260923-rapid-evidence.json",
      "phipps": {
        "status": "complete",
        "measured": 36,
        "required": 36,
        "visible_finals": 36,
        "score": 4.66,
        "gate_count": 3,
        "judge_false_positive_count": 1,
        "hard_gate_failure_ids": [
          "coding_safe_command_help",
          "knowledge_current_hours",
          "memory_current_turn_over_store"
        ],
        "note": "One gated macOS command was independently replayed successfully against native system tools. Canonical score and all recorded gates remain unchanged; the diagnostic is not a rescore."
      },
      "livebench": {
        "status": "partial",
        "answered": 17,
        "required": 90,
        "judgments_resolved": 0,
        "score": null,
        "note": "Stopped after an upstream stream error. Successful retained answers remain partial; no full-suite score or fake zeros."
      },
      "bfcl": {
        "status": "complete",
        "correct": 19,
        "required": 29,
        "score": 65.52,
        "official_full_comparable": false,
        "note": "Untouched BFCL component completed in a separate bounded continuation after the original upstream stream failure; original records are retained."
      },
      "performance": {
        "p50_s": 4.634825167013332,
        "p95_s": 37.64234487502836,
        "output_tokens_per_second": 62.73165835741057,
        "note": "Latency includes failed attempts; token rate uses usage-verified requests, includes reasoning, and is not native decode. Tokenizers and providers differ. Maximum request latency: 367.4801545829978."
      },
      "cost": {
        "generation_usd": 1.7813208,
        "note": "Conservative guard upper estimate, not an invoice; CNY numeric rates counted as USD without an FX discount. Held campaign liabilities for this model: $1.070514. Admissions are separate from benchmark generation."
      },
      "provenance": {
        "campaign_summary_sha256": "585934a33d84cfe41ad72fd1e5ad537e79e85540aa3ef0ef924e4a94a18859d3",
        "evidence_hashes": {
          "policy": "525ba9ebd57955ba95bc4616beaf791e1690a386978283b6edee2d9bf0c83309",
          "ledger": "096dc364ebdd48df4d90609562b930f835c535863f1de0760a41d98b6479fa48",
          "execution": "f8105db323e874995c1449edf33ed8fd3d3546100f6e922f7c176fa9cda2bd46",
          "bfcl_continuation": "abaafc2fd7e16d04a02ec95359e98674192ea5e446c941c90488c4b6efb7a141",
          "bfcl": "c8846e516188009c782fc43fbdaed7a1463389018141e1dfa9049aa101923c4a",
          "phipps": "62dc3dbfb041de00930865d7fb61b1f5dd82c3d5ea345e7a4946bb4b6639e8a6",
          "phipps_rows": "bbddb564314c93d23a6340f37d473176ca22c8491bbdbde6650ab138db6f2a50",
          "judge_diagnostic": "70290521c9f5cf434cd17eb464aa7c31fb53c9c0c6533abee09b9b9249ad0c92"
        }
      }
    },
    {
      "id": "capstan-deepseek-cloud-rapid-20260923",
      "campaign_id": "capstan-cloud-rapid-20260923",
      "campaign_status": "partial",
      "status": "partial",
      "name": "DeepSeek V4.1 Flash",
      "model_id": "deepseek-flash",
      "provider": "DeepSeek direct",
      "reasoning": "Native max; visible-final-only grading",
      "output_limit_tokens": 393216,
      "native_max_caveat": "Native maximum reasoning and output; 600-second request watchdog. Model-specific native profiles are not matched token-budget or local-runtime comparisons.",
      "eval_date": "2026-09-23",
      "rankable": false,
      "promotion": false,
      "overall_score": null,
      "receipt_url": "/data/external/capstan-deepseek-cloud-rapid-20260923-rapid-evidence.json",
      "phipps": {
        "status": "partial",
        "measured": 13,
        "required": 36,
        "visible_finals": 12,
        "score": null,
        "gate_count": 2,
        "judge_false_positive_count": 0,
        "hard_gate_failure_ids": [
          "interaction_direct_recommendation",
          "tool_web_conflict_synthesis"
        ],
        "note": "Incomplete product component: incomplete_generation. No composite or passing credit is inferred."
      },
      "livebench": {
        "status": "partial",
        "answered": 83,
        "required": 90,
        "judgments_resolved": 0,
        "score": null,
        "note": "Stopped at the preset component wall limit. Retained answers are incomplete coverage, not a full-suite score."
      },
      "bfcl": {
        "status": "complete",
        "correct": 21,
        "required": 29,
        "score": 72.41,
        "official_full_comparable": false,
        "note": null
      },
      "performance": {
        "p50_s": 3.8897426455514506,
        "p95_s": 84.02491012495011,
        "output_tokens_per_second": 215.76627287019534,
        "note": "Latency includes failed attempts; token rate uses usage-verified requests, includes reasoning, and is not native decode. Tokenizers and providers differ. Maximum request latency: 199.67178466683254."
      },
      "cost": {
        "generation_usd": 2.4688218,
        "note": "Conservative peak uncached token-price upper estimate, not an invoice. Held campaign liabilities for this model: $0.000000. Admissions are separate from benchmark generation."
      },
      "provenance": {
        "campaign_summary_sha256": "585934a33d84cfe41ad72fd1e5ad537e79e85540aa3ef0ef924e4a94a18859d3",
        "evidence_hashes": {
          "policy": "2ee6b4736fee05d652358701c28516ba011dffabf69a1db1640c57af4a2cb0e2",
          "ledger": "096dc364ebdd48df4d90609562b930f835c535863f1de0760a41d98b6479fa48",
          "execution": "fed19b32cdc2e0940ae75d4f095b5fff6b36c605b5dd55a9d4f22b459b512b91",
          "phipps": "a5adc7fee13668ef13c9d5c317975bd3df1e1a04ca1790c42b9113d97747f80c",
          "bfcl": "1fab3750e330b2dd6c8f68e0492d2ddc43280557173cb2cfcb2d9849de534e76",
          "phipps_rows": "c06f3c4409ecd839e9c22efd90f3ccc5700d8513cce58976b64266d4f9a37ff4"
        }
      }
    },
    {
      "id": "capstan-glm-cloud-rapid-recovery-20260923-001",
      "campaign_id": "capstan-cloud-rapid-recovery-20260923-001",
      "campaign_status": "partial",
      "status": "partial",
      "name": "GLM-5.3 Flash recovered Rapid",
      "model_id": "z-ai/glm-5.3-flash",
      "provider": "Z.AI via OpenRouter FP8",
      "reasoning": "Native max; visible-final-only grading",
      "output_limit_tokens": 131072,
      "native_max_caveat": "Native maximum reasoning and output; 600-second request watchdog. Model-specific native profiles are not matched token-budget or local-runtime comparisons. Official Z.AI FP8 route pinned; no provider fallback. Direct account lacked prepaid balance.",
      "eval_date": "2026-09-23",
      "rankable": false,
      "promotion": false,
      "overall_score": null,
      "descriptive_overall": null,
      "receipt_url": "/data/external/capstan-glm-cloud-rapid-recovery-20260923-001-rapid-evidence.json",
      "phipps": {
        "status": "complete",
        "measured": 36,
        "required": 36,
        "visible_finals": 36,
        "score": 4.27,
        "gate_count": 6,
        "judge_false_positive_count": 0,
        "hard_gate_failure_ids": [
          "instruction_exact_json",
          "interaction_direct_recommendation",
          "knowledge_current_hours",
          "memory_current_turn_over_store",
          "memory_long_context_distractor",
          "tool_web_service_snapshot"
        ],
        "note": null
      },
      "livebench": {
        "status": "partial",
        "answered": 45,
        "required": 90,
        "judgments_resolved": 0,
        "score": null,
        "note": "Recovery generation is incomplete; observed nonempty answers are not a scored component."
      },
      "bfcl": {
        "status": "complete",
        "correct": 20,
        "required": 29,
        "score": 68.97,
        "official_full_comparable": false,
        "note": null
      },
      "performance": {
        "p50_s": 8.070995167130604,
        "p95_s": 123.89314516703598,
        "output_tokens_per_second": 31.428744950053037,
        "note": "Latency includes all benchmark attempts, including failed tails; rates use only usage-verified output/end-to-end tokens. Admissions are excluded."
      },
      "cost": {
        "generation_usd": 0.10771374,
        "note": "Cumulative successor ledger view: predecessor and recovery records counted once; unsettled holds remain liabilities. Not an invoice."
      },
      "provenance": {
        "campaign_summary_sha256": "ef0a7b8f644e51ac1b9d480dec1275703deae09d0bb55aaca1993f13fd7a04fc",
        "evidence_hashes": {
          "request_records": "20e9d46d4662699d9e36ed96c7f977136c1e013b9071b543234a14f754cbd5a7",
          "recovery_contract": "e64666df84aa84563a5804bcf06966914f37d72155de54a9da927ffabd4e8898",
          "predecessor_verified_summary": "f196720017113d8f74921e4b97e9bd982a2da757e07af5ded4ac6f38944a636a",
          "linked_ledger": "955c5da2f6505c2f4c4ae4bb5a9f126d2123b83bae884f3a132266380fc29d63",
          "admission": "937059fe11cec63c03e9a1aa0c5ed0b95516538a2bebf29bdd03abf30e8c279e",
          "livebench_execution": "a734ed0006e307453d0795b6c7d2e9c6abf346e2e393c16c14aa12715c546d24"
        }
      }
    },
    {
      "id": "capstan-qwen-cloud-rapid-recovery-20260923-001",
      "campaign_id": "capstan-cloud-rapid-recovery-20260923-001",
      "campaign_status": "partial",
      "status": "partial",
      "name": "Qwen3.8 Flash recovered Rapid",
      "model_id": "qwen3.8-flash",
      "provider": "Alibaba direct US",
      "reasoning": "Native xhigh; visible-final-only grading",
      "output_limit_tokens": 131072,
      "native_max_caveat": "Native maximum reasoning and output; 600-second request watchdog. Model-specific native profiles are not matched token-budget or local-runtime comparisons. Hosted Qwen3.8 Flash is a derivative, not verified identical to local Qwen3.8-Flash-Next weights.",
      "eval_date": "2026-09-23",
      "rankable": false,
      "promotion": false,
      "overall_score": null,
      "descriptive_overall": null,
      "receipt_url": "/data/external/capstan-qwen-cloud-rapid-recovery-20260923-001-rapid-evidence.json",
      "phipps": {
        "status": "complete",
        "measured": 36,
        "required": 36,
        "visible_finals": 36,
        "score": 4.66,
        "gate_count": 3,
        "judge_false_positive_count": 1,
        "hard_gate_failure_ids": [
          "coding_safe_command_help",
          "knowledge_current_hours",
          "memory_current_turn_over_store"
        ],
        "note": "One gated macOS command was independently replayed successfully against native system tools. Canonical score and all recorded gates remain unchanged; the diagnostic is not a rescore."
      },
      "livebench": {
        "status": "partial",
        "answered": 40,
        "required": 90,
        "judgments_resolved": 0,
        "score": null,
        "note": "Recovery generation is incomplete; observed nonempty answers are not a scored component."
      },
      "bfcl": {
        "status": "complete",
        "correct": 19,
        "required": 29,
        "score": 65.52,
        "official_full_comparable": false,
        "note": "Untouched BFCL component completed in a separate bounded continuation after the original upstream stream failure; original records are retained."
      },
      "performance": {
        "p50_s": 5.205537791829556,
        "p95_s": 65.7594309579581,
        "output_tokens_per_second": 62.99104602280238,
        "note": "Latency includes all benchmark attempts, including failed tails; rates use only usage-verified output/end-to-end tokens. Admissions are excluded."
      },
      "cost": {
        "generation_usd": 1.9185704,
        "note": "Cumulative successor ledger view: predecessor and recovery records counted once; unsettled holds remain liabilities. Not an invoice."
      },
      "provenance": {
        "campaign_summary_sha256": "ef0a7b8f644e51ac1b9d480dec1275703deae09d0bb55aaca1993f13fd7a04fc",
        "evidence_hashes": {
          "request_records": "b7201ba1389d52bb0e9bbc3cc66323fe856510fcde462eb52159f7aea84ba685",
          "recovery_contract": "e64666df84aa84563a5804bcf06966914f37d72155de54a9da927ffabd4e8898",
          "predecessor_verified_summary": "f196720017113d8f74921e4b97e9bd982a2da757e07af5ded4ac6f38944a636a",
          "linked_ledger": "955c5da2f6505c2f4c4ae4bb5a9f126d2123b83bae884f3a132266380fc29d63",
          "admission": "767c8fd5bd8fe0382e058800241157f10c5a2bae015a98b85b0fe5b688b311a3",
          "livebench_execution": "c74040da0b5773a1c2d547ad7b1c74eeb529953a673b03d30de819209e523145"
        }
      }
    },
    {
      "id": "capstan-deepseek-cloud-rapid-recovery-20260923-001",
      "campaign_id": "capstan-cloud-rapid-recovery-20260923-001",
      "campaign_status": "partial",
      "status": "complete",
      "name": "DeepSeek V4.1 Flash recovered Rapid",
      "model_id": "deepseek-flash",
      "provider": "DeepSeek direct",
      "reasoning": "Native max; visible-final-only grading",
      "output_limit_tokens": 393216,
      "native_max_caveat": "Native maximum reasoning and output; 600-second request watchdog. Model-specific native profiles are not matched token-budget or local-runtime comparisons.",
      "eval_date": "2026-09-23",
      "rankable": false,
      "promotion": false,
      "overall_score": null,
      "descriptive_overall": 88.4,
      "receipt_url": "/data/external/capstan-deepseek-cloud-rapid-recovery-20260923-001-rapid-evidence.json",
      "phipps": {
        "status": "complete",
        "measured": 36,
        "required": 36,
        "visible_finals": 36,
        "score": 4.62,
        "gate_count": 3,
        "judge_false_positive_count": 0,
        "hard_gate_failure_ids": [
          "interaction_direct_recommendation",
          "knowledge_current_hours",
          "tool_web_conflict_synthesis"
        ],
        "note": "Fresh 36x1 recovery result; the predecessor's distinct missing-final semantic failure remains preserved and is not erased by this pass."
      },
      "livebench": {
        "status": "complete",
        "answered": 90,
        "required": 90,
        "judgments_resolved": 90,
        "score": 92.24,
        "note": null
      },
      "bfcl": {
        "status": "complete",
        "correct": 21,
        "required": 29,
        "score": 72.41,
        "official_full_comparable": false,
        "note": null
      },
      "performance": {
        "p50_s": 3.0629829788813367,
        "p95_s": 79.87019641604275,
        "output_tokens_per_second": 215.54257931848778,
        "note": "Latency includes all benchmark attempts, including failed tails; rates use only usage-verified output/end-to-end tokens. Admissions are excluded."
      },
      "cost": {
        "generation_usd": 2.5994637,
        "note": "Cumulative successor ledger view: predecessor and recovery records counted once; unsettled holds remain liabilities. Not an invoice."
      },
      "provenance": {
        "campaign_summary_sha256": "ef0a7b8f644e51ac1b9d480dec1275703deae09d0bb55aaca1993f13fd7a04fc",
        "evidence_hashes": {
          "request_records": "e48bf2a8de61d75afa3b70ed4fe7ae89a3a094b14f0eddcf09e0f359d2e75dc0",
          "recovery_contract": "e64666df84aa84563a5804bcf06966914f37d72155de54a9da927ffabd4e8898",
          "predecessor_verified_summary": "f196720017113d8f74921e4b97e9bd982a2da757e07af5ded4ac6f38944a636a",
          "linked_ledger": "955c5da2f6505c2f4c4ae4bb5a9f126d2123b83bae884f3a132266380fc29d63",
          "admission": "1405968997d9a9b9bf8f94a706aa841b697c6a35053161f4dcb754bea61708d8",
          "livebench_execution": "ebf644b038ac684655194d49a5ac5fcf2c4fdaa68afc2caa00fbb3184e64fedc",
          "phipps_execution": "7287be465d3ac511bc1af1c4b43c78af5014f25dd2f121fd15af18bbfd665c0e"
        }
      }
    },
    {
      "id": "six-cloud-mimo26pro-v41-20260928",
      "campaign_id": "capstan-six-cloud-native-v41-20260928",
      "campaign_status": "partial",
      "status": "blocked",
      "name": "Mimo v2.6-pro",
      "model_id": "mimo-v2.6-pro",
      "provider": "Xiaomi MiMo direct API (not admitted)",
      "reasoning": "No admitted benchmark profile. Proposed native thinking only; controls/history not claimed verified.",
      "output_limit_tokens": 131072,
      "native_max_caveat": "Authenticated admission or preparation blocked before benchmark: Only Token Plan credential found; automated general benchmark scripts forbidden by current MiMo Token Plan terms. No pay-as-you-go credential found in checked authorized stores.",
      "eval_date": "2026-09-28",
      "rankable": false,
      "promotion": false,
      "overall_score": null,
      "descriptive_overall": null,
      "receipt_url": "/data/external/six-cloud-mimo26pro-v41-20260928-rapid-evidence.json",
      "phipps": {
        "status": "blocked",
        "measured": 0,
        "required": 36,
        "visible_finals": 0,
        "score": null,
        "gate_count": 0,
        "judge_false_positive_count": 0,
        "hard_gate_failure_ids": [],
        "note": "Authenticated admission or preparation blocked before benchmark: Only Token Plan credential found; automated general benchmark scripts forbidden by current MiMo Token Plan terms. No pay-as-you-go credential found in checked authorized stores."
      },
      "livebench": {
        "status": "blocked",
        "answered": 0,
        "required": 90,
        "judgments_resolved": 0,
        "score": null,
        "note": "Authenticated admission or preparation blocked before benchmark: Only Token Plan credential found; automated general benchmark scripts forbidden by current MiMo Token Plan terms. No pay-as-you-go credential found in checked authorized stores."
      },
      "bfcl": {
        "status": "blocked",
        "correct": null,
        "required": 29,
        "score": null,
        "official_full_comparable": false,
        "note": "Authenticated admission or preparation blocked before benchmark: Only Token Plan credential found; automated general benchmark scripts forbidden by current MiMo Token Plan terms. No pay-as-you-go credential found in checked authorized stores."
      },
      "performance": {
        "p50_s": null,
        "p95_s": null,
        "output_tokens_per_second": null,
        "note": "No benchmark requests; admission timing is not model performance."
      },
      "cost": {
        "generation_usd": 0.0,
        "note": "Protocol estimates only, not invoice settlement. Additional conservative unresolved liability $0. No benchmark cost claimed."
      },
      "provenance": {
        "campaign_summary_sha256": "a57d8a60a490638d6b8aedb89d9a7a860ff9f5ed1c30dd43d569c68a8401918a",
        "evidence_hashes": {
          "intake": "7b0f835c98670fe9f4d07539d86d0e82c7f6669be9ba39ed570e5672da6779e8",
          "route_blockers": "d097d7c5aa13342a38f03f39e700969eaa88eb9d40ba2c2688e80dcf272de4d0",
          "attempt-0001-status": "9adb5d7c4665bea5be4d3812d5801968a686c0527dfcf5370600a09e7d24b350"
        }
      }
    },
    {
      "id": "six-cloud-glm53-v41-20260928",
      "campaign_id": "capstan-six-cloud-native-v41-20260928",
      "campaign_status": "partial",
      "status": "blocked",
      "name": "GLM-5.3",
      "model_id": "glm-5.3",
      "provider": "Z.AI direct API (not admitted)",
      "reasoning": "No admitted benchmark profile. Proposed native thinking only; controls/history not claimed verified.",
      "output_limit_tokens": 131072,
      "native_max_caveat": "Direct account unfunded (Flash admission HTTP429/code1113); billing sign-in unavailable. OpenRouter alternative not admitted: permission for frozen injection fixture unestablished under Terms section7. No quality result.",
      "eval_date": "2026-09-28",
      "rankable": false,
      "promotion": false,
      "overall_score": null,
      "descriptive_overall": null,
      "receipt_url": "/data/external/six-cloud-glm53-v41-20260928-rapid-evidence.json",
      "phipps": {
        "status": "blocked",
        "measured": 0,
        "required": 36,
        "visible_finals": 0,
        "score": null,
        "gate_count": 0,
        "judge_false_positive_count": 0,
        "hard_gate_failure_ids": [],
        "note": "Direct account unfunded (Flash admission HTTP429/code1113); billing sign-in unavailable. OpenRouter alternative not admitted: permission for frozen injection fixture unestablished under Terms section7. No quality result."
      },
      "livebench": {
        "status": "blocked",
        "answered": 0,
        "required": 90,
        "judgments_resolved": 0,
        "score": null,
        "note": "Direct account unfunded (Flash admission HTTP429/code1113); billing sign-in unavailable. OpenRouter alternative not admitted: permission for frozen injection fixture unestablished under Terms section7. No quality result."
      },
      "bfcl": {
        "status": "blocked",
        "correct": null,
        "required": 29,
        "score": null,
        "official_full_comparable": false,
        "note": "Direct account unfunded (Flash admission HTTP429/code1113); billing sign-in unavailable. OpenRouter alternative not admitted: permission for frozen injection fixture unestablished under Terms section7. No quality result."
      },
      "performance": {
        "p50_s": null,
        "p95_s": null,
        "output_tokens_per_second": null,
        "note": "No benchmark requests; admission timing is not model performance."
      },
      "cost": {
        "generation_usd": 0.0,
        "note": "Protocol estimates only, not invoice settlement. Additional conservative unresolved liability $0. No benchmark cost claimed."
      },
      "provenance": {
        "campaign_summary_sha256": "a57d8a60a490638d6b8aedb89d9a7a860ff9f5ed1c30dd43d569c68a8401918a",
        "evidence_hashes": {
          "intake": "7b0f835c98670fe9f4d07539d86d0e82c7f6669be9ba39ed570e5672da6779e8",
          "route_blockers": "d097d7c5aa13342a38f03f39e700969eaa88eb9d40ba2c2688e80dcf272de4d0"
        }
      }
    },
    {
      "id": "six-cloud-glm53flash-v41-20260928",
      "campaign_id": "capstan-six-cloud-native-v41-20260928",
      "campaign_status": "partial",
      "status": "blocked",
      "name": "GLM 5.3-Flash",
      "model_id": "glm-5.3-flash",
      "provider": "Z.AI direct API",
      "reasoning": "No admitted benchmark profile. Proposed native thinking only; controls/history not claimed verified.",
      "output_limit_tokens": 131072,
      "native_max_caveat": "Direct account unfunded (Flash admission HTTP429/code1113); billing sign-in unavailable. OpenRouter alternative not admitted: permission for frozen injection fixture unestablished under Terms section7. No quality result.",
      "eval_date": "2026-09-28",
      "rankable": false,
      "promotion": false,
      "overall_score": null,
      "descriptive_overall": null,
      "receipt_url": "/data/external/six-cloud-glm53flash-v41-20260928-rapid-evidence.json",
      "phipps": {
        "status": "blocked",
        "measured": 0,
        "required": 36,
        "visible_finals": 0,
        "score": null,
        "gate_count": 0,
        "judge_false_positive_count": 0,
        "hard_gate_failure_ids": [],
        "note": "Direct account unfunded (Flash admission HTTP429/code1113); billing sign-in unavailable. OpenRouter alternative not admitted: permission for frozen injection fixture unestablished under Terms section7. No quality result."
      },
      "livebench": {
        "status": "blocked",
        "answered": 0,
        "required": 90,
        "judgments_resolved": 0,
        "score": null,
        "note": "Direct account unfunded (Flash admission HTTP429/code1113); billing sign-in unavailable. OpenRouter alternative not admitted: permission for frozen injection fixture unestablished under Terms section7. No quality result."
      },
      "bfcl": {
        "status": "blocked",
        "correct": null,
        "required": 29,
        "score": null,
        "official_full_comparable": false,
        "note": "Direct account unfunded (Flash admission HTTP429/code1113); billing sign-in unavailable. OpenRouter alternative not admitted: permission for frozen injection fixture unestablished under Terms section7. No quality result."
      },
      "performance": {
        "p50_s": null,
        "p95_s": null,
        "output_tokens_per_second": null,
        "note": "No benchmark requests; admission timing is not model performance."
      },
      "cost": {
        "generation_usd": 0.0,
        "note": "Protocol estimates only, not invoice settlement. Additional conservative unresolved liability $0.06680935. No benchmark cost claimed."
      },
      "provenance": {
        "campaign_summary_sha256": "a57d8a60a490638d6b8aedb89d9a7a860ff9f5ed1c30dd43d569c68a8401918a",
        "evidence_hashes": {
          "intake": "7b0f835c98670fe9f4d07539d86d0e82c7f6669be9ba39ed570e5672da6779e8",
          "route_blockers": "d097d7c5aa13342a38f03f39e700969eaa88eb9d40ba2c2688e80dcf272de4d0",
          "attempt-0001-status": "0a0900738449c8394e41fd364dcbd64ab33db42f8eb1bd489f257e1a70d1f9ad",
          "attempt-0001-policy": "b0b80b64f215a99c97dca408a28314cc09e020b248a9fe76379f963a837500c2"
        }
      }
    },
    {
      "id": "six-cloud-qwen38flashnext-v41-20260928",
      "campaign_id": "capstan-six-cloud-native-v41-20260928",
      "campaign_status": "partial",
      "status": "blocked",
      "name": "Qwen 3.8-flash-next",
      "model_id": "Qwen/Qwen3.8-Flash-Next",
      "provider": "Featherless API candidate (not admitted)",
      "reasoning": "No admitted benchmark profile. Proposed native thinking only; controls/history not claimed verified.",
      "output_limit_tokens": 32768,
      "native_max_caveat": "Exact Featherless catalog identity found but no authorized credentials; native xhigh/history forwarding unverified. Provider32768 ceiling is below publisher ceiling. Related hosted Flash not substituted.",
      "eval_date": "2026-09-28",
      "rankable": false,
      "promotion": false,
      "overall_score": null,
      "descriptive_overall": null,
      "receipt_url": "/data/external/six-cloud-qwen38flashnext-v41-20260928-rapid-evidence.json",
      "phipps": {
        "status": "blocked",
        "measured": 0,
        "required": 36,
        "visible_finals": 0,
        "score": null,
        "gate_count": 0,
        "judge_false_positive_count": 0,
        "hard_gate_failure_ids": [],
        "note": "Exact Featherless catalog identity found but no authorized credentials; native xhigh/history forwarding unverified. Provider32768 ceiling is below publisher ceiling. Related hosted Flash not substituted."
      },
      "livebench": {
        "status": "blocked",
        "answered": 0,
        "required": 90,
        "judgments_resolved": 0,
        "score": null,
        "note": "Exact Featherless catalog identity found but no authorized credentials; native xhigh/history forwarding unverified. Provider32768 ceiling is below publisher ceiling. Related hosted Flash not substituted."
      },
      "bfcl": {
        "status": "blocked",
        "correct": null,
        "required": 29,
        "score": null,
        "official_full_comparable": false,
        "note": "Exact Featherless catalog identity found but no authorized credentials; native xhigh/history forwarding unverified. Provider32768 ceiling is below publisher ceiling. Related hosted Flash not substituted."
      },
      "performance": {
        "p50_s": null,
        "p95_s": null,
        "output_tokens_per_second": null,
        "note": "No benchmark requests; admission timing is not model performance."
      },
      "cost": {
        "generation_usd": 0.0,
        "note": "Protocol estimates only, not invoice settlement. Additional conservative unresolved liability $0. No benchmark cost claimed."
      },
      "provenance": {
        "campaign_summary_sha256": "a57d8a60a490638d6b8aedb89d9a7a860ff9f5ed1c30dd43d569c68a8401918a",
        "evidence_hashes": {
          "intake": "7b0f835c98670fe9f4d07539d86d0e82c7f6669be9ba39ed570e5672da6779e8",
          "route_blockers": "d097d7c5aa13342a38f03f39e700969eaa88eb9d40ba2c2688e80dcf272de4d0"
        }
      }
    },
    {
      "id": "six-cloud-deepseek41flash-v41-context-recovered-20260928",
      "campaign_id": "capstan-six-cloud-native-v41-20260928",
      "campaign_status": "partial",
      "status": "complete",
      "name": "deepseek v4.1-flash",
      "model_id": "deepseek-flash",
      "provider": "DeepSeek direct API",
      "reasoning": "Native thinking; {'effort': 'max', 'exclude': False}; full native assistant reasoning history preserved. Sampling 1.0/0.95; Codex gpt-5.6-sol xhigh Phipps judge.",
      "output_limit_tokens": 393216,
      "native_max_caveat": "Direct DeepSeek native max effort;393216 output ceiling. Phipps36 and LiveBench90 retained unchanged. BFCL28 original trajectories plus one infrastructure-only recovery. Explicit dual transport-source provenance; exact prior request-wire equivalence verified. No prompt, sampling, reasoning-history or output-budget change. One-run Screen, not a matched deployment ranking; four Phipps gates remain. One auxiliary admission probe hit the stream byte guard; its unscored output and full billing reserve are retained separately. Fresh protocol admission passed before the BFCL recovery.",
      "eval_date": "2026-09-28",
      "rankable": false,
      "promotion": false,
      "overall_score": null,
      "descriptive_overall": 86.9565,
      "receipt_url": "/data/external/six-cloud-deepseek41flash-v41-context-recovered-20260928-rapid-evidence.json",
      "phipps": {
        "status": "complete",
        "measured": 36,
        "required": 36,
        "visible_finals": 36,
        "score": 4.54,
        "gate_count": 4,
        "judge_false_positive_count": 1,
        "hard_gate_failure_ids": [
          "interaction_direct_recommendation",
          "memory_current_turn_over_store",
          "knowledge_current_hours",
          "coding_safe_command_help"
        ],
        "note": "Revised v4.1: hard failures receive rubric floor1/5, zero normalized credit. Complete measurement is not qualification. One coding row has verified macOS-portability and deletion-keyword false-positive diagnostics; canonical score/gate remain unchanged. Immediate-versus-recursive scope remains unresolved."
      },
      "livebench": {
        "status": "complete",
        "answered": 90,
        "required": 90,
        "judgments_resolved": 90,
        "score": 92.46,
        "note": "Frozen90-question Screen; official scorer with source-bound same-model gpt-5-mini OpenRouter/OpenAI equivalence fallback. Not Full."
      },
      "bfcl": {
        "status": "complete",
        "correct": 20,
        "required": 29,
        "score": 68.97,
        "official_full_comparable": false,
        "note": "Frozen29 Rapid.28 original trajectories preserved byte-for-byte; one infrastructure-missing trajectory recovered with context-accounting-only correction. Native scorer and raw verification passed. No valid answer rerolled; not official Full."
      },
      "performance": {
        "p50_s": 3.6276559790130705,
        "p95_s": 81.12183983391151,
        "output_tokens_per_second": 217.211210322945,
        "note": "API request end-to-end, not local decode. Includes original benchmark requests and failed-trajectory prefix plus exact-missing-case recovery; admissions excluded. Max 341.098s; 286requests. Throughput includes hidden reasoning divided by summed request wall time. p95 nearest-rank."
      },
      "cost": {
        "generation_usd": 5.4245064,
        "note": "Cumulative DeepSeek peak-rate estimate including admissions, retired diagnostic attempts and recovery; not invoice settlement. Additional unresolved liability $0.4743936 from a stopped auxiliary admission probe remains reserved under explicit operator authorization. Shared LiveBench equivalence-scorer cost is separate."
      },
      "provenance": {
        "campaign_summary_sha256": "a57d8a60a490638d6b8aedb89d9a7a860ff9f5ed1c30dd43d569c68a8401918a",
        "evidence_hashes": {
          "policy": "c80a4198f1f9ccdc693791a465551a5d80a64137edc97aef8f742c71e58c022f",
          "admission": "f730a2291bb5fa672a3ddf1929e6fb4b1e501ec7134a24a923555d5477be972d",
          "state": "6dfd46d759d10bc9dd3122782bad6666614b58acda59e11fdf6c473d61ec68f0",
          "request_records": "a51f3082b3a33b7e0dade7ca0eb3c8ca3b90c8b5773dd27eacc09cdefe9b1642",
          "revised_policy": "cfb245a2b4ef81cf06ae46c443f4d11315110f3a6e684b7c8cfcc94fadca98ac",
          "runner": "b5e81749562bab5ecb29eb53159e4d7c6e69683a4a2a3027228f2984d4cb73b9",
          "phipps": "8bd6591e780cfe72c29b10d90057e4c263b5bb1a24820214f9447c32ca7c6d4b",
          "livebench": "abe9a9e2e7107826ee3e3c03a48fbfb55fa7e5296098aca4ca2b6b50bed3036a",
          "coding_diagnostic": "3aec94a4a70b63f4f5a1260dc499b112d851571ed1a7ae72d18e0c643babafc2",
          "bfcl": "138dc6f56c2cd31ee34bc5a082e663dbdfed2d344ea4215c60311fa899680bdd",
          "context_recovery": "b160725d2f836ee3b30ce40880549b07280c3b9dbe064b0aa8a88ac1c1c65386",
          "context_recovery_plan": "73ff4c6ebbddcd81c3d009905a187c4d69c0369a2851acfcff2fc1780d40d72e",
          "context_recovery_admission": "f344597ae520087b762a6e431d9e96015d75820b5397d23f9a2aaaab220f5600",
          "context_regressions": "e2edb08412cbf89bc90c5eb3041e6823fc97158b1fc579ae641cd2fe6fa076f7",
          "recovery_spend_authorization": "e09b0036fce52bc34ff6bfb102137b36faf8a06d0f8e8bade3f7bf16586b5278"
        }
      }
    },
    {
      "id": "six-cloud-mimo26flash-v41-20260928",
      "campaign_id": "capstan-six-cloud-native-v41-20260928",
      "campaign_status": "partial",
      "status": "blocked",
      "name": "mimo v2.6flash",
      "model_id": "mimo-v2.6-flash",
      "provider": "Xiaomi MiMo direct API (not admitted)",
      "reasoning": "No admitted benchmark profile. Proposed native thinking only; controls/history not claimed verified.",
      "output_limit_tokens": 131072,
      "native_max_caveat": "Authenticated admission or preparation blocked before benchmark: Only Token Plan credential found; automated general benchmark scripts forbidden by current MiMo Token Plan terms. No pay-as-you-go credential found in checked authorized stores.",
      "eval_date": "2026-09-28",
      "rankable": false,
      "promotion": false,
      "overall_score": null,
      "descriptive_overall": null,
      "receipt_url": "/data/external/six-cloud-mimo26flash-v41-20260928-rapid-evidence.json",
      "phipps": {
        "status": "blocked",
        "measured": 0,
        "required": 36,
        "visible_finals": 0,
        "score": null,
        "gate_count": 0,
        "judge_false_positive_count": 0,
        "hard_gate_failure_ids": [],
        "note": "Authenticated admission or preparation blocked before benchmark: Only Token Plan credential found; automated general benchmark scripts forbidden by current MiMo Token Plan terms. No pay-as-you-go credential found in checked authorized stores."
      },
      "livebench": {
        "status": "blocked",
        "answered": 0,
        "required": 90,
        "judgments_resolved": 0,
        "score": null,
        "note": "Authenticated admission or preparation blocked before benchmark: Only Token Plan credential found; automated general benchmark scripts forbidden by current MiMo Token Plan terms. No pay-as-you-go credential found in checked authorized stores."
      },
      "bfcl": {
        "status": "blocked",
        "correct": null,
        "required": 29,
        "score": null,
        "official_full_comparable": false,
        "note": "Authenticated admission or preparation blocked before benchmark: Only Token Plan credential found; automated general benchmark scripts forbidden by current MiMo Token Plan terms. No pay-as-you-go credential found in checked authorized stores."
      },
      "performance": {
        "p50_s": null,
        "p95_s": null,
        "output_tokens_per_second": null,
        "note": "No benchmark requests; admission timing is not model performance."
      },
      "cost": {
        "generation_usd": 0.0,
        "note": "Protocol estimates only, not invoice settlement. Additional conservative unresolved liability $0. No benchmark cost claimed."
      },
      "provenance": {
        "campaign_summary_sha256": "a57d8a60a490638d6b8aedb89d9a7a860ff9f5ed1c30dd43d569c68a8401918a",
        "evidence_hashes": {
          "intake": "7b0f835c98670fe9f4d07539d86d0e82c7f6669be9ba39ed570e5672da6779e8",
          "route_blockers": "d097d7c5aa13342a38f03f39e700969eaa88eb9d40ba2c2688e80dcf272de4d0",
          "attempt-0001-status": "257f6af7a6999596c2b1141d586b8792bb03cc729b7c87c3011bc50b03e3802d"
        }
      }
    },
    {
      "evaluation_profile": {
        "policy_version": "capstan-cloud-eval-v4.1-or154",
        "count": 35,
        "runs": 1,
        "ids": [
          "interaction_direct_recommendation",
          "interaction_empathic_reply",
          "interaction_firm_email",
          "interaction_meeting_summary",
          "interaction_casual_tone",
          "interaction_answer_first",
          "tool_web_service_snapshot",
          "tool_web_restraint_provided_text",
          "tool_calculator_exact",
          "tool_calculator_restraint",
          "tool_memory_project_recall",
          "tool_memory_restraint",
          "tool_web_release_lookup",
          "instruction_exact_json",
          "instruction_four_bullets",
          "instruction_exact_transformation",
          "instruction_three_lines_regex",
          "instruction_distractor_override",
          "instruction_bounded_status",
          "memory_correction_inheritance",
          "memory_preference_persistence",
          "memory_long_context_distractor",
          "memory_durable_decision_tool",
          "memory_current_turn_over_store",
          "knowledge_false_premise",
          "knowledge_missing_attachment",
          "knowledge_current_hours",
          "knowledge_stable_fact_restraint",
          "knowledge_source_priority",
          "knowledge_insufficient_age",
          "coding_debug_off_by_one",
          "coding_review_retry_loop",
          "coding_safe_command_help",
          "coding_implement_chunks",
          "coding_security_review"
        ],
        "excluded_ids": [
          "tool_web_conflict_synthesis"
        ],
        "original_count": 36,
        "full36_comparable": false,
        "official_full_comparable": false,
        "ids_sha256": "66191964fa6bf2ce9304dfedf1b3f74a8c0263732819ec1ec6e48d4003feac85"
      },
      "id": "six-cloud-mimo26pro-or-or154-20260928",
      "campaign_id": "capstan-openrouter-native-v41-154-20260928",
      "campaign_status": "partial",
      "status": "partial",
      "name": "Mimo v2.6-pro",
      "model_id": "xiaomi/mimo-v2.6-pro",
      "provider": "Xiaomi FP8 via OpenRouter (official pinned)",
      "reasoning": "Native {\"enabled\":true,\"exclude\":false}; T1.0/top_p0.95; reasoning and reasoning_details history retained. Phipps Codex gpt-5.6-sol xhigh judge.",
      "output_limit_tokens": 131072,
      "native_max_caveat": "Official FP8 OpenRouter route, fallback disabled. Native thinking enabled; binary effort;131072provider output ceiling. Full reasoning/reasoning_details retained at the OpenRouter interface; not proof of hidden vendor clear_thinking flags. T1/top_p.95 requested; vendor thinking-mode sampling behavior may override. API end-to-end speed is not local decode. Distinct154-case profile, no direct/full155comparability claim. Timeout-only continuation, not sampling/reasoning retuning. Exact original Phipps35/BFCL29 preserved; LiveBench dual-source provenance, unchanged generation wire. Recovery stopped without another retry; remaining coverage is unscored missingness, no Overall.",
      "eval_date": "2026-09-28",
      "rankable": false,
      "promotion": false,
      "overall_score": null,
      "descriptive_overall": null,
      "receipt_url": "/data/external/six-cloud-mimo26pro-or-or154-20260928-rapid-evidence.json",
      "phipps": {
        "status": "complete",
        "measured": 35,
        "required": 35,
        "visible_finals": 35,
        "score": 4.59,
        "gate_count": 3,
        "judge_false_positive_count": 0,
        "hard_gate_failure_ids": [
          "interaction_direct_recommendation",
          "knowledge_current_hours",
          "coding_safe_command_help"
        ],
        "note": "Versioned35-task cohort, excluding tool_web_conflict_synthesis. Each hard gate receives floor1/5, zero normalized credit; completion is not qualification."
      },
      "livebench": {
        "status": "partial",
        "answered": 43,
        "required": 90,
        "judgments_resolved": 0,
        "score": null,
        "note": "Frozen90 Screen, not Full. 40 byte-bound retained answers plus3 disjoint missing-ID-only successor answers; 0 valid native judgments. Same generation/scorer controls;1800/1850s operational watch successor, original600/650s STOP and liability preserved. Measured outcomes include41 visible finals and2 proven physical-ceiling semantic zeros (no passing credit)."
      },
      "bfcl": {
        "status": "complete",
        "correct": 23,
        "required": 29,
        "score": 79.31,
        "official_full_comparable": false,
        "note": "Frozen29 Rapid, not Full. Native failure-only scorer artifacts and all trajectories raw-revalidated."
      },
      "performance": {
        "p50_s": 6.006593625061214,
        "p95_s": 53.22736954106949,
        "output_tokens_per_second": 62.80636955300809,
        "note": "Cloud API end-to-end, not local decode or Capstan prediction. Original+single-recovery: 183 benchmark requests, max1566.472s, 5 requests>=600s. Nearest-rank p95 includes all failed requests; token rate includes reasoning but excludes unknown usage, never unknown latency."
      },
      "cost": {
        "generation_usd": 0.3509368722,
        "note": "Provider-reported usage charges for this OR lane including7protocol calls and initial benign access probe, not an invoice. Additional unresolved reserve $0.236296785 retained. Shared scorer costs reported separately in campaign totals. Includes single missing-ID recovery; original uncertain reserves unchanged and counted once. Not invoice settlement."
      },
      "provenance": {
        "campaign_summary_sha256": "17a431f8041c4a612cc5bc9018e577cb0d84ab1b65b3b3983d0f34d8821aefcc",
        "evidence_hashes": {
          "policy": "3ff485ddb678f01a6b5e4eb011a0b51301db4e04ef0f4d6815189efb11adb09f",
          "admission": "3f782576cb5ed894c6e35c9c43145772e71c56c8f79e23043497d2b54de9f6cc",
          "execution": "a548e01485f9946feb57926e4d57f0a827c4129856a592e04e0207ccd054f0b3",
          "source_seal": "86b88c5d98ae4fcc72a6106eb495055286a46f1667adc1cd6dfdb4b3379dd7c6",
          "state": "ba43e7b8efc1af39dd8235f6295c3731d81765da40bae7e9433ad34795c8206f",
          "request_records": "fc63a2ae79cf035e7ac9c5867e7c151ab3fde01d8672b51e696af67755aa0054",
          "scope_authorization": "07a35d7e0e2641dbfd9e3bdcc220a0168e8b403eb372a6307be6939116d2afa3",
          "phipps": "5d5c662eb5166b0244525393115a6fcb4a5d0501fc3d38c4d8c70ad664c9fcff",
          "bfcl": "7ae13f9cfd2a29b79f80ff0d300a3b6116a75ee3b75502f63f7831b8509f5988",
          "preserved_stop": "8f16457da9b63dae80e7df8fb1b4690896a46991b5d896fa03060b8e10a8f204",
          "recovery_plan": "461991b76c3ef014bb257dc2eaaed98a8f70775ce159a48f0a1465a28b0be480",
          "recovery_status": "4ae509a94e8078f37e960e4c1f1e0629037490d0ffcb131f72a11cd4a0f718f3",
          "recovery_preflight": "0884451c707801ecf2d95ddad46db1ffb8da5e54308134e7b848e8fe434dd867",
          "recovery_wire_equivalence": "fbb9b698da039c7760051cd46cd24c03a0744d51a94a54be4eb86147af8cfd66",
          "recovery_requests": "2fb1126048d375c08829db333070ebfb5c026bc9c422b6c213c0ad8ab5667cf1",
          "recovery_raw_files": "e516bc604eb7ac2305dc900291279d56ab370003881824eb998cbaa544ee9367",
          "overnight_authority": "fa57a44bcb18a2f78507f0de5070082e79974c72b822509e07df813554f3c0bc",
          "independent_recovery_auditor": "d14d84da09f22a368fea61002d017cd2c632f9e44f406e901509e5729a8345a5"
        }
      }
    },
    {
      "evaluation_profile": {
        "policy_version": "capstan-cloud-eval-v4.1-or154",
        "count": 35,
        "runs": 1,
        "ids": [
          "interaction_direct_recommendation",
          "interaction_empathic_reply",
          "interaction_firm_email",
          "interaction_meeting_summary",
          "interaction_casual_tone",
          "interaction_answer_first",
          "tool_web_service_snapshot",
          "tool_web_restraint_provided_text",
          "tool_calculator_exact",
          "tool_calculator_restraint",
          "tool_memory_project_recall",
          "tool_memory_restraint",
          "tool_web_release_lookup",
          "instruction_exact_json",
          "instruction_four_bullets",
          "instruction_exact_transformation",
          "instruction_three_lines_regex",
          "instruction_distractor_override",
          "instruction_bounded_status",
          "memory_correction_inheritance",
          "memory_preference_persistence",
          "memory_long_context_distractor",
          "memory_durable_decision_tool",
          "memory_current_turn_over_store",
          "knowledge_false_premise",
          "knowledge_missing_attachment",
          "knowledge_current_hours",
          "knowledge_stable_fact_restraint",
          "knowledge_source_priority",
          "knowledge_insufficient_age",
          "coding_debug_off_by_one",
          "coding_review_retry_loop",
          "coding_safe_command_help",
          "coding_implement_chunks",
          "coding_security_review"
        ],
        "excluded_ids": [
          "tool_web_conflict_synthesis"
        ],
        "original_count": 36,
        "full36_comparable": false,
        "official_full_comparable": false,
        "ids_sha256": "66191964fa6bf2ce9304dfedf1b3f74a8c0263732819ec1ec6e48d4003feac85"
      },
      "id": "six-cloud-glm53-or-or154-20260928",
      "campaign_id": "capstan-openrouter-native-v41-154-20260928",
      "campaign_status": "partial",
      "status": "partial",
      "name": "GLM-5.3",
      "model_id": "z-ai/glm-5.3",
      "provider": "Z.AI FP8 via OpenRouter (official pinned)",
      "reasoning": "Native {\"effort\":\"max\",\"exclude\":false}; T1.0/top_p0.95; reasoning and reasoning_details history retained. Phipps Codex gpt-5.6-sol xhigh judge.",
      "output_limit_tokens": 131072,
      "native_max_caveat": "Official FP8 OpenRouter route, fallback disabled. Native thinking max effort requested;131072provider output ceiling. Full reasoning/reasoning_details retained at the OpenRouter interface; not proof of hidden vendor clear_thinking flags. T1/top_p.95 requested; vendor thinking-mode sampling behavior may override. API end-to-end speed is not local decode. Distinct154-case profile, no direct/full155comparability claim. Timeout-only continuation, not sampling/reasoning retuning. Exact original Phipps35/BFCL29 preserved; LiveBench dual-source provenance, unchanged generation wire. Recovery stopped without another retry; remaining coverage is unscored missingness, no Overall.",
      "eval_date": "2026-09-28",
      "rankable": false,
      "promotion": false,
      "overall_score": null,
      "descriptive_overall": null,
      "receipt_url": "/data/external/six-cloud-glm53-or-or154-20260928-rapid-evidence.json",
      "phipps": {
        "status": "complete",
        "measured": 35,
        "required": 35,
        "visible_finals": 35,
        "score": 4.38,
        "gate_count": 5,
        "judge_false_positive_count": 0,
        "hard_gate_failure_ids": [
          "interaction_direct_recommendation",
          "interaction_answer_first",
          "tool_web_service_snapshot",
          "knowledge_current_hours",
          "coding_safe_command_help"
        ],
        "note": "Versioned35-task cohort, excluding tool_web_conflict_synthesis. Each hard gate receives floor1/5, zero normalized credit; completion is not qualification."
      },
      "livebench": {
        "status": "partial",
        "answered": 58,
        "required": 90,
        "judgments_resolved": 0,
        "score": null,
        "note": "Frozen90 Screen, not Full. 40 byte-bound retained answers plus18 disjoint missing-ID-only successor answers; 0 valid native judgments. Same generation/scorer controls;1800/1850s operational watch successor, original600/650s STOP and liability preserved. Measured outcomes include58 visible finals and0 proven physical-ceiling semantic zeros (no passing credit)."
      },
      "bfcl": {
        "status": "complete",
        "correct": 21,
        "required": 29,
        "score": 72.41,
        "official_full_comparable": false,
        "note": "Frozen29 Rapid, not Full. Native failure-only scorer artifacts and all trajectories raw-revalidated."
      },
      "performance": {
        "p50_s": 8.652725500054657,
        "p95_s": 248.3504512081854,
        "output_tokens_per_second": 49.27564299060438,
        "note": "Cloud API end-to-end, not local decode or Capstan prediction. Original+single-recovery: 205 benchmark requests, max1090.493s, 5 requests>=600s. Nearest-rank p95 includes all failed requests; token rate includes reasoning but excludes unknown usage, never unknown latency."
      },
      "cost": {
        "generation_usd": 2.76168268,
        "note": "Provider-reported usage charges for this OR lane including7protocol calls and initial benign access probe, not an invoice. Additional unresolved reserve $1.1826978 retained. Shared scorer costs reported separately in campaign totals. Includes single missing-ID recovery; original uncertain reserves unchanged and counted once. Not invoice settlement."
      },
      "provenance": {
        "campaign_summary_sha256": "17a431f8041c4a612cc5bc9018e577cb0d84ab1b65b3b3983d0f34d8821aefcc",
        "evidence_hashes": {
          "policy": "c8730deabafd0c709287899aafec6d34548b7042783148fe2e83f69594eea453",
          "admission": "f1ab77d0ac7bbbad8d6159ddee68e0e72be97b7bd1cdda1e99d4b9cc09a73e13",
          "execution": "b3b801114a1783cdbba855b923a09c52cfe35e734aa2128ec5053a188e025296",
          "source_seal": "0a041845915f2dff261300425eacbcd8a1ab8d2ce6fea582829650a76f3f0fec",
          "state": "7067f3f2c824d4d8738fdf382060499a3c61b7a30bee502cb84ad1e7412010b0",
          "request_records": "3ea28760785c16958fa88ac62446e00fe5e67cfe3814a470d4418cb2ce109f01",
          "scope_authorization": "07a35d7e0e2641dbfd9e3bdcc220a0168e8b403eb372a6307be6939116d2afa3",
          "phipps": "4685b54d9de54d395b2a5752773a271c599b67183719584431df201fd3fa6bbf",
          "bfcl": "0676db0d8eddeb479ca3fce32b04cc3c7ab71420fdbd9691e30195825493d1c6",
          "preserved_stop": "8f16457da9b63dae80e7df8fb1b4690896a46991b5d896fa03060b8e10a8f204",
          "recovery_plan": "ecf0c96f51e7cb805d1e28eadcafc3f76b2b6b6c37d48348ab31ebf34673fcb2",
          "recovery_status": "0120be3aa781f21721462235d9282531544b87f1400d1f96a3cdf9477d4635cd",
          "recovery_preflight": "a8c192097337ce0bd8df2178d0e59ac659d231124130b77e038315468cb2bdea",
          "recovery_wire_equivalence": "8cce3da1baf2c842ed7209ffc433c7407d761dd6620b96a0b61e41fc024f4b53",
          "recovery_requests": "59d53b1854e88eb1329c188ad04f6c5622da6fefaea49399905f6f30287bbc58",
          "recovery_raw_files": "49408b0ce33bc9db938f79f9a70f7aabb3e82db4cdca303b556dcd79f77cdb5e",
          "overnight_authority": "fa57a44bcb18a2f78507f0de5070082e79974c72b822509e07df813554f3c0bc",
          "independent_recovery_auditor": "d14d84da09f22a368fea61002d017cd2c632f9e44f406e901509e5729a8345a5"
        }
      }
    },
    {
      "evaluation_profile": {
        "policy_version": "capstan-cloud-eval-v4.1-or154",
        "count": 35,
        "runs": 1,
        "ids": [
          "interaction_direct_recommendation",
          "interaction_empathic_reply",
          "interaction_firm_email",
          "interaction_meeting_summary",
          "interaction_casual_tone",
          "interaction_answer_first",
          "tool_web_service_snapshot",
          "tool_web_restraint_provided_text",
          "tool_calculator_exact",
          "tool_calculator_restraint",
          "tool_memory_project_recall",
          "tool_memory_restraint",
          "tool_web_release_lookup",
          "instruction_exact_json",
          "instruction_four_bullets",
          "instruction_exact_transformation",
          "instruction_three_lines_regex",
          "instruction_distractor_override",
          "instruction_bounded_status",
          "memory_correction_inheritance",
          "memory_preference_persistence",
          "memory_long_context_distractor",
          "memory_durable_decision_tool",
          "memory_current_turn_over_store",
          "knowledge_false_premise",
          "knowledge_missing_attachment",
          "knowledge_current_hours",
          "knowledge_stable_fact_restraint",
          "knowledge_source_priority",
          "knowledge_insufficient_age",
          "coding_debug_off_by_one",
          "coding_review_retry_loop",
          "coding_safe_command_help",
          "coding_implement_chunks",
          "coding_security_review"
        ],
        "excluded_ids": [
          "tool_web_conflict_synthesis"
        ],
        "original_count": 36,
        "full36_comparable": false,
        "official_full_comparable": false,
        "ids_sha256": "66191964fa6bf2ce9304dfedf1b3f74a8c0263732819ec1ec6e48d4003feac85"
      },
      "id": "six-cloud-glm53flash-or-or154-20260928",
      "campaign_id": "capstan-openrouter-native-v41-154-20260928",
      "campaign_status": "partial",
      "status": "partial",
      "name": "GLM 5.3-Flash",
      "model_id": "z-ai/glm-5.3-flash",
      "provider": "Z.AI FP8 via OpenRouter (official pinned)",
      "reasoning": "Native {\"effort\":\"max\",\"exclude\":false}; T1.0/top_p0.95; reasoning and reasoning_details history retained. Phipps Codex gpt-5.6-sol xhigh judge.",
      "output_limit_tokens": 131072,
      "native_max_caveat": "Official FP8 OpenRouter route, fallback disabled. Native thinking max effort requested;131072provider output ceiling. Full reasoning/reasoning_details retained at the OpenRouter interface; not proof of hidden vendor clear_thinking flags. T1/top_p.95 requested; vendor thinking-mode sampling behavior may override. API end-to-end speed is not local decode. Distinct154-case profile, no direct/full155comparability claim. Timeout-only continuation, not sampling/reasoning retuning. Exact original Phipps35/BFCL29 preserved; LiveBench dual-source provenance, unchanged generation wire. Recovery stopped without another retry; remaining coverage is unscored missingness, no Overall.",
      "eval_date": "2026-09-28",
      "rankable": false,
      "promotion": false,
      "overall_score": null,
      "descriptive_overall": null,
      "receipt_url": "/data/external/six-cloud-glm53flash-or-or154-20260928-rapid-evidence.json",
      "phipps": {
        "status": "complete",
        "measured": 35,
        "required": 35,
        "visible_finals": 35,
        "score": 4.28,
        "gate_count": 6,
        "judge_false_positive_count": 0,
        "hard_gate_failure_ids": [
          "interaction_direct_recommendation",
          "interaction_answer_first",
          "instruction_exact_json",
          "instruction_four_bullets",
          "memory_current_turn_over_store",
          "knowledge_current_hours"
        ],
        "note": "Versioned35-task cohort, excluding tool_web_conflict_synthesis. Each hard gate receives floor1/5, zero normalized credit; completion is not qualification."
      },
      "livebench": {
        "status": "partial",
        "answered": 45,
        "required": 90,
        "judgments_resolved": 0,
        "score": null,
        "note": "Frozen90 Screen, not Full. 45 byte-bound retained answers plus0 disjoint missing-ID-only successor answers; 0 valid native judgments. Same generation/scorer controls;1800/1850s operational watch successor, original600/650s STOP and liability preserved. Measured outcomes include45 visible finals and0 proven physical-ceiling semantic zeros (no passing credit)."
      },
      "bfcl": {
        "status": "complete",
        "correct": 23,
        "required": 29,
        "score": 79.31,
        "official_full_comparable": false,
        "note": "Frozen29 Rapid, not Full. Native failure-only scorer artifacts and all trajectories raw-revalidated."
      },
      "performance": {
        "p50_s": 7.77932808350306,
        "p95_s": 88.50906562502496,
        "output_tokens_per_second": 40.189864821342276,
        "note": "Cloud API end-to-end, not local decode or Capstan prediction. Original+single-recovery: 180 benchmark requests, max1800.108s, 2 requests>=600s. Nearest-rank p95 includes all failed requests; token rate includes reasoning but excludes unknown usage, never unknown latency."
      },
      "cost": {
        "generation_usd": 0.09318356,
        "note": "Provider-reported usage charges for this OR lane including7protocol calls and initial benign access probe, not an invoice. Additional unresolved reserve $0.13479170 retained. Shared scorer costs reported separately in campaign totals. Includes single missing-ID recovery; original uncertain reserves unchanged and counted once. Not invoice settlement."
      },
      "provenance": {
        "campaign_summary_sha256": "17a431f8041c4a612cc5bc9018e577cb0d84ab1b65b3b3983d0f34d8821aefcc",
        "evidence_hashes": {
          "policy": "5c28cbf64691eca62c888acba5f1c3d9a514c264793e2b7ca2cf2fcfff5e2ed3",
          "admission": "559a309a1d26952ff5745a452aab670ae08912e0999f64fb2b6d9e2f727b4294",
          "execution": "5221fdb23cf4f64085d8517ec449aa5501e711040eacfe55586398c756b1a786",
          "source_seal": "61a597532cc9d9b28012f087e080215e651c96788cb9be9d78531e6d0f6af7e3",
          "state": "ae1cebe3577de8a0079a7c32a90a75548ba9d972890dd1ad0e51cfa2c146933b",
          "request_records": "a2e2e63a0041b0607bc856eac2154daee9e977156547c45f15aecd6860b52da0",
          "scope_authorization": "07a35d7e0e2641dbfd9e3bdcc220a0168e8b403eb372a6307be6939116d2afa3",
          "phipps": "2e21d96075c14ed8ff4ac5fa04101837b7a5b75fab89704fc3464e3c4e3d4915",
          "bfcl": "ee2443b4693cf400f5586bd3e1d50ceb80a27482298e6973acbe3aa53d5af5ef",
          "preserved_stop": "8f16457da9b63dae80e7df8fb1b4690896a46991b5d896fa03060b8e10a8f204",
          "recovery_plan": "8c9d3cfc8e171ddd24dedc677a661944efa4542af097dd118e2917233fa4a3c9",
          "recovery_status": "7ed5a86657b4b797770018d386f562e36c954f7526803095da0e052536537179",
          "recovery_preflight": "aae8521159680048b51ae89e9db92a4977c1748bc5ad945d2bf30d2c15c569fb",
          "recovery_wire_equivalence": "46c847a92ec1d85366ed820da8b4f7ae31bad20ae406ee66aa0753a1284ed0f6",
          "recovery_requests": "14652bf40fb2964f1683779818956aebf16ec65addc410084906c05c2f8ee752",
          "recovery_raw_files": "88ece1d256b87694f76736a61ccd29fdf9b1ca5b9237957fc133a0e492ec5b00",
          "overnight_authority": "fa57a44bcb18a2f78507f0de5070082e79974c72b822509e07df813554f3c0bc",
          "independent_recovery_auditor": "d14d84da09f22a368fea61002d017cd2c632f9e44f406e901509e5729a8345a5"
        }
      }
    },
    {
      "evaluation_profile": {
        "policy_version": "capstan-cloud-eval-v4.1-or154",
        "count": 35,
        "runs": 1,
        "ids": [
          "interaction_direct_recommendation",
          "interaction_empathic_reply",
          "interaction_firm_email",
          "interaction_meeting_summary",
          "interaction_casual_tone",
          "interaction_answer_first",
          "tool_web_service_snapshot",
          "tool_web_restraint_provided_text",
          "tool_calculator_exact",
          "tool_calculator_restraint",
          "tool_memory_project_recall",
          "tool_memory_restraint",
          "tool_web_release_lookup",
          "instruction_exact_json",
          "instruction_four_bullets",
          "instruction_exact_transformation",
          "instruction_three_lines_regex",
          "instruction_distractor_override",
          "instruction_bounded_status",
          "memory_correction_inheritance",
          "memory_preference_persistence",
          "memory_long_context_distractor",
          "memory_durable_decision_tool",
          "memory_current_turn_over_store",
          "knowledge_false_premise",
          "knowledge_missing_attachment",
          "knowledge_current_hours",
          "knowledge_stable_fact_restraint",
          "knowledge_source_priority",
          "knowledge_insufficient_age",
          "coding_debug_off_by_one",
          "coding_review_retry_loop",
          "coding_safe_command_help",
          "coding_implement_chunks",
          "coding_security_review"
        ],
        "excluded_ids": [
          "tool_web_conflict_synthesis"
        ],
        "original_count": 36,
        "full36_comparable": false,
        "official_full_comparable": false,
        "ids_sha256": "66191964fa6bf2ce9304dfedf1b3f74a8c0263732819ec1ec6e48d4003feac85"
      },
      "id": "six-cloud-mimo26flash-or-or154-20260928",
      "campaign_id": "capstan-openrouter-native-v41-154-20260928",
      "campaign_status": "partial",
      "status": "partial",
      "name": "mimo v2.6flash",
      "model_id": "xiaomi/mimo-v2.6-flash",
      "provider": "Xiaomi FP8 via OpenRouter (official pinned)",
      "reasoning": "Native {\"enabled\":true,\"exclude\":false}; T1.0/top_p0.95; reasoning and reasoning_details history retained. Phipps Codex gpt-5.6-sol xhigh judge.",
      "output_limit_tokens": 131072,
      "native_max_caveat": "Official FP8 OpenRouter route, fallback disabled. Native thinking enabled; binary effort;131072provider output ceiling. Full reasoning/reasoning_details retained at the OpenRouter interface; not proof of hidden vendor clear_thinking flags. T1/top_p.95 requested; vendor thinking-mode sampling behavior may override. API end-to-end speed is not local decode. Distinct154-case profile, no direct/full155comparability claim. Timeout-only continuation, not sampling/reasoning retuning. Exact original Phipps35/BFCL29 preserved; LiveBench dual-source provenance, unchanged generation wire. Recovery stopped without another retry; remaining coverage is unscored missingness, no Overall.",
      "eval_date": "2026-09-28",
      "rankable": false,
      "promotion": false,
      "overall_score": null,
      "descriptive_overall": null,
      "receipt_url": "/data/external/six-cloud-mimo26flash-or-or154-20260928-rapid-evidence.json",
      "phipps": {
        "status": "complete",
        "measured": 35,
        "required": 35,
        "visible_finals": 35,
        "score": 4.2,
        "gate_count": 7,
        "judge_false_positive_count": 0,
        "hard_gate_failure_ids": [
          "interaction_direct_recommendation",
          "interaction_answer_first",
          "instruction_exact_json",
          "instruction_four_bullets",
          "memory_long_context_distractor",
          "knowledge_current_hours",
          "coding_safe_command_help"
        ],
        "note": "Versioned35-task cohort, excluding tool_web_conflict_synthesis. Each hard gate receives floor1/5, zero normalized credit; completion is not qualification."
      },
      "livebench": {
        "status": "partial",
        "answered": 86,
        "required": 90,
        "judgments_resolved": 0,
        "score": null,
        "note": "Frozen90 Screen, not Full. 41 byte-bound retained answers plus45 disjoint missing-ID-only successor answers; 0 valid native judgments. Same generation/scorer controls;1800/1850s operational watch successor, original600/650s STOP and liability preserved. Measured outcomes include86 visible finals and0 proven physical-ceiling semantic zeros (no passing credit)."
      },
      "bfcl": {
        "status": "complete",
        "correct": 24,
        "required": 29,
        "score": 82.76,
        "official_full_comparable": false,
        "note": "Frozen29 Rapid, not Full. Native failure-only scorer artifacts and all trajectories raw-revalidated."
      },
      "performance": {
        "p50_s": 6.218061417108402,
        "p95_s": 453.371159292059,
        "output_tokens_per_second": 66.18560838171221,
        "note": "Cloud API end-to-end, not local decode or Capstan prediction. Original+single-recovery: 227 benchmark requests, max1800.118s, 8 requests>=600s. Nearest-rank p95 includes all failed requests; token rate includes reasoning but excludes unknown usage, never unknown latency."
      },
      "cost": {
        "generation_usd": 0.2834610352,
        "note": "Provider-reported usage charges for this OR lane including7protocol calls and initial benign access probe, not an invoice. Additional unresolved reserve $0.07637952 retained. Shared scorer costs reported separately in campaign totals. Includes single missing-ID recovery; original uncertain reserves unchanged and counted once. Not invoice settlement."
      },
      "provenance": {
        "campaign_summary_sha256": "17a431f8041c4a612cc5bc9018e577cb0d84ab1b65b3b3983d0f34d8821aefcc",
        "evidence_hashes": {
          "policy": "cf5b4216f1d61b9b5f5ab8134a1d9aa58c78cea785ba3d8cfe23082ff36020b0",
          "admission": "1e6d8941ce0843ecde7089fe1b2013b97c70ab2aba2e6722e93ed6117b9c33d1",
          "execution": "9b3c956639970a6f83996ba7fb10f2103e4e48e95911980024e22e68ddd23fc4",
          "source_seal": "6167b4349f5d83d25078072362c0dbc64497256f6fc0fbe49678df135a2bcafa",
          "state": "b1c2d00dc8b8a4961efe687b4be45c2cc900ba9b3f2cbdc9429306639a342bbd",
          "request_records": "add81f7044cf85f36d4b4bb290de27c6dfa489ea5a496631317bb1c2ce66bdc0",
          "scope_authorization": "07a35d7e0e2641dbfd9e3bdcc220a0168e8b403eb372a6307be6939116d2afa3",
          "phipps": "578e46b130f552cd66c8f80a165b12a3f97b66f5a43dc4f5be30263ce52404e2",
          "bfcl": "0cd7192cf0e36dff3af7bb41240602f5ed77f760edef0b6c102dcb3eee8466c9",
          "preserved_stop": "8f16457da9b63dae80e7df8fb1b4690896a46991b5d896fa03060b8e10a8f204",
          "recovery_plan": "4d899646753e5478c88adc8a9b7d1da3430301e8a083917abeb97026b582a442",
          "recovery_status": "2c486a44e62fbe3d7ef5150abeb363abd7d84d8f163458cbf7dbf2e9ad2f85dd",
          "recovery_preflight": "2066200d4e6d039af0bed90aee710af91fe5018c68f2f6455c3f53dae3f41056",
          "recovery_wire_equivalence": "703afa1c390aba27e7b544e5127ea25a9741524a6c06f3dcb9c8634c71afb501",
          "recovery_requests": "169f425552ae2672b42a1d0b96e9a044a6b2be2ab73dfcb3df55ee9a1ee926af",
          "recovery_raw_files": "3be72967c3e5939bd8c9bc45524398da4f3a9bd6a3210b4338ed1b98a4164fe9",
          "overnight_authority": "fa57a44bcb18a2f78507f0de5070082e79974c72b822509e07df813554f3c0bc",
          "independent_recovery_auditor": "d14d84da09f22a368fea61002d017cd2c632f9e44f406e901509e5729a8345a5"
        }
      }
    },
    {
      "evaluation_profile": {
        "policy_version": "capstan-cloud-eval-v4.1-or154",
        "count": 35,
        "runs": 1,
        "ids": [
          "interaction_direct_recommendation",
          "interaction_empathic_reply",
          "interaction_firm_email",
          "interaction_meeting_summary",
          "interaction_casual_tone",
          "interaction_answer_first",
          "tool_web_service_snapshot",
          "tool_web_restraint_provided_text",
          "tool_calculator_exact",
          "tool_calculator_restraint",
          "tool_memory_project_recall",
          "tool_memory_restraint",
          "tool_web_release_lookup",
          "instruction_exact_json",
          "instruction_four_bullets",
          "instruction_exact_transformation",
          "instruction_three_lines_regex",
          "instruction_distractor_override",
          "instruction_bounded_status",
          "memory_correction_inheritance",
          "memory_preference_persistence",
          "memory_long_context_distractor",
          "memory_durable_decision_tool",
          "memory_current_turn_over_store",
          "knowledge_false_premise",
          "knowledge_missing_attachment",
          "knowledge_current_hours",
          "knowledge_stable_fact_restraint",
          "knowledge_source_priority",
          "knowledge_insufficient_age",
          "coding_debug_off_by_one",
          "coding_review_retry_loop",
          "coding_safe_command_help",
          "coding_implement_chunks",
          "coding_security_review"
        ],
        "excluded_ids": [
          "tool_web_conflict_synthesis"
        ],
        "original_count": 36,
        "full36_comparable": false,
        "official_full_comparable": false,
        "ids_sha256": "66191964fa6bf2ce9304dfedf1b3f74a8c0263732819ec1ec6e48d4003feac85"
      },
      "id": "six-cloud-mimo26flash-or-or154-last4-20260929",
      "campaign_id": "capstan-openrouter-native-v41-154-20260928",
      "campaign_status": "partial",
      "status": "complete",
      "name": "mimo v2.6flash",
      "model_id": "xiaomi/mimo-v2.6-flash",
      "provider": "Xiaomi FP8 via OpenRouter (official pinned)",
      "reasoning": "Native {\"enabled\":true,\"exclude\":false}; T1.0/top_p0.95; reasoning and reasoning_details history retained. Phipps Codex gpt-5.6-sol xhigh judge.",
      "output_limit_tokens": 131072,
      "native_max_caveat": "Official FP8 OpenRouter route, fallback disabled. Native thinking enabled; binary effort;131072provider output ceiling. Full reasoning/reasoning_details retained at the OpenRouter interface; not proof of hidden vendor clear_thinking flags. T1/top_p.95 requested; vendor thinking-mode sampling behavior may override. API end-to-end speed is not local decode. Distinct154-case profile, no direct/full155comparability claim. Separately authorized exact-four continuation; original+first-recovery+last-four provenance. Original STOPs, timeout liabilities and settled repetition-truncation failure preserved; generation/scorer unchanged. All154 measured; seven Phipps hard gates still reject qualification.",
      "eval_date": "2026-09-29",
      "rankable": false,
      "promotion": false,
      "overall_score": null,
      "descriptive_overall": 80.1725,
      "receipt_url": "/data/external/six-cloud-mimo26flash-or-or154-last4-20260929-rapid-evidence.json",
      "phipps": {
        "status": "complete",
        "measured": 35,
        "required": 35,
        "visible_finals": 35,
        "score": 4.2,
        "gate_count": 7,
        "judge_false_positive_count": 0,
        "hard_gate_failure_ids": [
          "interaction_direct_recommendation",
          "interaction_answer_first",
          "instruction_exact_json",
          "instruction_four_bullets",
          "memory_long_context_distractor",
          "knowledge_current_hours",
          "coding_safe_command_help"
        ],
        "note": "Versioned35-task cohort, excluding tool_web_conflict_synthesis. Each hard gate receives floor1/5, zero normalized credit; completion is not qualification."
      },
      "livebench": {
        "status": "complete",
        "answered": 90,
        "required": 90,
        "judgments_resolved": 90,
        "score": 79.31,
        "note": "Frozen90 Screen, not Full.41 original+45 first-recovery+4 exact-missing-ID continuation outcomes;90 valid native judgments. 90 visible finals,0 physical-ceiling zeros. Retained answer objects unchanged;3600/3650s operational watches only. Prior repetition-truncation no-final remains a failure, not a semantic zero."
      },
      "bfcl": {
        "status": "complete",
        "correct": 24,
        "required": 29,
        "score": 82.76,
        "official_full_comparable": false,
        "note": "Frozen29 Rapid, not Full. Native failure-only scorer artifacts and all trajectories raw-revalidated."
      },
      "performance": {
        "p50_s": 6.223477874882519,
        "p95_s": 453.371159292059,
        "output_tokens_per_second": 65.87931054883937,
        "note": "Cloud API end-to-end, not local decode or Capstan prediction. Three source roots: 231 benchmark calls; max1800.118s. Nearest-rank p95 includes failed requests. Rate sums usage-known completion tokens including reasoning over usage-known wall time; unknown usage excluded from rate, never latency."
      },
      "cost": {
        "generation_usd": 0.2923531352,
        "note": "Provider-reported charges across all three source roots plus original admission/access. Unresolved reserve $0.07637952 preserved and counted once; includes prior settled no-final failure. Shared scorer billed separately. Not invoice settlement."
      },
      "provenance": {
        "campaign_summary_sha256": "efa3a02d7ebdcf13df042ae2603022090d77c7cf13528429ccfb90dcb5f936c4",
        "evidence_hashes": {
          "policy": "cf5b4216f1d61b9b5f5ab8134a1d9aa58c78cea785ba3d8cfe23082ff36020b0",
          "admission": "1e6d8941ce0843ecde7089fe1b2013b97c70ab2aba2e6722e93ed6117b9c33d1",
          "execution": "9b3c956639970a6f83996ba7fb10f2103e4e48e95911980024e22e68ddd23fc4",
          "source_seal": "6167b4349f5d83d25078072362c0dbc64497256f6fc0fbe49678df135a2bcafa",
          "state": "b1c2d00dc8b8a4961efe687b4be45c2cc900ba9b3f2cbdc9429306639a342bbd",
          "request_records": "add81f7044cf85f36d4b4bb290de27c6dfa489ea5a496631317bb1c2ce66bdc0",
          "scope_authorization": "07a35d7e0e2641dbfd9e3bdcc220a0168e8b403eb372a6307be6939116d2afa3",
          "phipps": "578e46b130f552cd66c8f80a165b12a3f97b66f5a43dc4f5be30263ce52404e2",
          "bfcl": "0cd7192cf0e36dff3af7bb41240602f5ed77f760edef0b6c102dcb3eee8466c9",
          "preserved_stop": "8f16457da9b63dae80e7df8fb1b4690896a46991b5d896fa03060b8e10a8f204",
          "recovery_plan": "4d899646753e5478c88adc8a9b7d1da3430301e8a083917abeb97026b582a442",
          "recovery_status": "2c486a44e62fbe3d7ef5150abeb363abd7d84d8f163458cbf7dbf2e9ad2f85dd",
          "recovery_preflight": "2066200d4e6d039af0bed90aee710af91fe5018c68f2f6455c3f53dae3f41056",
          "recovery_wire_equivalence": "703afa1c390aba27e7b544e5127ea25a9741524a6c06f3dcb9c8634c71afb501",
          "recovery_requests": "169f425552ae2672b42a1d0b96e9a044a6b2be2ab73dfcb3df55ee9a1ee926af",
          "recovery_raw_files": "3be72967c3e5939bd8c9bc45524398da4f3a9bd6a3210b4338ed1b98a4164fe9",
          "overnight_authority": "fa57a44bcb18a2f78507f0de5070082e79974c72b822509e07df813554f3c0bc",
          "independent_recovery_auditor": "d14d84da09f22a368fea61002d017cd2c632f9e44f406e901509e5729a8345a5",
          "last4_plan": "984be67f8d8e82439bca3a866e8f88ea1037b35218380cf82c50ac807e35d5d9",
          "last4_status": "03fdeff5b220cf5e79ba1460b95b7549863ff069b14e4fb91e0ec64efaa7fd33",
          "last4_preflight": "aae7461532887626e0b05eb5073db527068aca9928b6b4e77d032699c71120be",
          "last4_scorer_preflight": "0e95e1baaa5bb6aa29852c3720599d8eac04de208536a59dd8ee6da5258e3dae",
          "last4_wire_equivalence": "c1f1f5ce85a810fae682c6de307a46591289c82e4dfcf6665938d1cad8c45f54",
          "last4_raw_files": "9287f48ac2bfcc6b141d48a88fa67ac3b93ed43adeb9d89270c5a1eebf1c4510",
          "multisource_answer_bindings": "b1c718bf37cc065e4d8c3bf1fe90990471b7820d79444bea29f5e374087d5659",
          "last4_auditor": "7426b984b01e05302b1a2a62fb11acff923dc55fa3feb5f9a863d9a1a8618232",
          "last4_authority": "9bffef44170cbf2fd0eded33621f82fd25e41be14dd4bfe51688dd9838ba2803",
          "cumulative_request_records": "ba83e77fbfbe22badf8b1b5ab5d1c19602ef261fa4e9730789a09e968eb2a0ee"
        }
      }
    },
    {
      "evaluation_profile": {
        "policy_version": "capstan-cloud-eval-v4.1-or154",
        "count": 35,
        "runs": 1,
        "ids": [
          "interaction_direct_recommendation",
          "interaction_empathic_reply",
          "interaction_firm_email",
          "interaction_meeting_summary",
          "interaction_casual_tone",
          "interaction_answer_first",
          "tool_web_service_snapshot",
          "tool_web_restraint_provided_text",
          "tool_calculator_exact",
          "tool_calculator_restraint",
          "tool_memory_project_recall",
          "tool_memory_restraint",
          "tool_web_release_lookup",
          "instruction_exact_json",
          "instruction_four_bullets",
          "instruction_exact_transformation",
          "instruction_three_lines_regex",
          "instruction_distractor_override",
          "instruction_bounded_status",
          "memory_correction_inheritance",
          "memory_preference_persistence",
          "memory_long_context_distractor",
          "memory_durable_decision_tool",
          "memory_current_turn_over_store",
          "knowledge_false_premise",
          "knowledge_missing_attachment",
          "knowledge_current_hours",
          "knowledge_stable_fact_restraint",
          "knowledge_source_priority",
          "knowledge_insufficient_age",
          "coding_debug_off_by_one",
          "coding_review_retry_loop",
          "coding_safe_command_help",
          "coding_implement_chunks",
          "coding_security_review"
        ],
        "excluded_ids": [
          "tool_web_conflict_synthesis"
        ],
        "original_count": 36,
        "full36_comparable": false,
        "official_full_comparable": false,
        "ids_sha256": "66191964fa6bf2ce9304dfedf1b3f74a8c0263732819ec1ec6e48d4003feac85"
      },
      "id": "six-cloud-glm53-or-or154-last32-20260929",
      "campaign_id": "capstan-openrouter-native-v41-154-20260928",
      "campaign_status": "partial",
      "status": "complete",
      "name": "GLM-5.3",
      "model_id": "z-ai/glm-5.3",
      "provider": "Z.AI FP8 via OpenRouter (official pinned)",
      "reasoning": "Native {\"effort\":\"max\",\"exclude\":false}; T1.0/top_p0.95; reasoning and reasoning_details history retained. Phipps Codex gpt-5.6-sol xhigh judge.",
      "output_limit_tokens": 131072,
      "native_max_caveat": "Official FP8 OpenRouter route, fallback disabled. Native thinking max effort requested;131072provider output ceiling. Full reasoning/reasoning_details retained at the OpenRouter interface; not proof of hidden vendor clear_thinking flags. T1/top_p.95 requested; vendor thinking-mode sampling behavior may override. API end-to-end speed is not local decode. Distinct154-case profile, no direct/full155comparability claim. Separately authorized exact-thirty-two continuation; original+first-recovery+last-thirty-two provenance. Original STOPs and both timeout liabilities preserved; generation/scorer unchanged. All154 measured; five Phipps hard gates still reject qualification.",
      "eval_date": "2026-09-29",
      "rankable": false,
      "promotion": false,
      "overall_score": null,
      "descriptive_overall": 83.9395,
      "receipt_url": "/data/external/six-cloud-glm53-or-or154-last32-20260929-rapid-evidence.json",
      "phipps": {
        "status": "complete",
        "measured": 35,
        "required": 35,
        "visible_finals": 35,
        "score": 4.38,
        "gate_count": 5,
        "judge_false_positive_count": 0,
        "hard_gate_failure_ids": [
          "interaction_direct_recommendation",
          "interaction_answer_first",
          "tool_web_service_snapshot",
          "knowledge_current_hours",
          "coding_safe_command_help"
        ],
        "note": "Versioned35-task cohort, excluding tool_web_conflict_synthesis. Each hard gate receives floor1/5, zero normalized credit; completion is not qualification."
      },
      "livebench": {
        "status": "complete",
        "answered": 90,
        "required": 90,
        "judgments_resolved": 90,
        "score": 88.08,
        "note": "Frozen90 Screen, not Full.40 original+18 first-recovery+32 exact-missing-ID continuation outcomes;90 valid native judgments. 89 visible finals,1 physical-ceiling zeros. Retained answer objects unchanged;3600/3650s operational watches only. Both prior timeout failures remain infrastructure missingness in their original attempts, never retroactive semantic zeros."
      },
      "bfcl": {
        "status": "complete",
        "correct": 21,
        "required": 29,
        "score": 72.41,
        "official_full_comparable": false,
        "note": "Frozen29 Rapid, not Full. Native failure-only scorer artifacts and all trajectories raw-revalidated."
      },
      "performance": {
        "p50_s": 9.61030679102987,
        "p95_s": 564.9649243340828,
        "output_tokens_per_second": 55.32427434633312,
        "note": "Cloud API end-to-end, not local decode or Capstan prediction. Three source roots: 237 benchmark calls; max2328.860s. Nearest-rank p95 includes failed requests. Rate sums usage-known completion tokens including reasoning over usage-known wall time; unknown usage excluded from rate, never latency."
      },
      "cost": {
        "generation_usd": 6.02594604,
        "note": "Provider-reported charges across all three source roots plus original admission/access. Unresolved reserve $1.1826978 preserved and counted once; includes original admission and timeout history. Shared scorer billed separately. Not invoice settlement."
      },
      "provenance": {
        "campaign_summary_sha256": "ef8a751f14bcd3fd964bbab67730d9f48738b455be6d047e43ba5c741584c15e",
        "evidence_hashes": {
          "policy": "c8730deabafd0c709287899aafec6d34548b7042783148fe2e83f69594eea453",
          "admission": "f1ab77d0ac7bbbad8d6159ddee68e0e72be97b7bd1cdda1e99d4b9cc09a73e13",
          "execution": "b3b801114a1783cdbba855b923a09c52cfe35e734aa2128ec5053a188e025296",
          "source_seal": "0a041845915f2dff261300425eacbcd8a1ab8d2ce6fea582829650a76f3f0fec",
          "state": "7067f3f2c824d4d8738fdf382060499a3c61b7a30bee502cb84ad1e7412010b0",
          "request_records": "3ea28760785c16958fa88ac62446e00fe5e67cfe3814a470d4418cb2ce109f01",
          "scope_authorization": "07a35d7e0e2641dbfd9e3bdcc220a0168e8b403eb372a6307be6939116d2afa3",
          "phipps": "4685b54d9de54d395b2a5752773a271c599b67183719584431df201fd3fa6bbf",
          "bfcl": "0676db0d8eddeb479ca3fce32b04cc3c7ab71420fdbd9691e30195825493d1c6",
          "preserved_stop": "8f16457da9b63dae80e7df8fb1b4690896a46991b5d896fa03060b8e10a8f204",
          "recovery_plan": "ecf0c96f51e7cb805d1e28eadcafc3f76b2b6b6c37d48348ab31ebf34673fcb2",
          "recovery_status": "0120be3aa781f21721462235d9282531544b87f1400d1f96a3cdf9477d4635cd",
          "recovery_preflight": "a8c192097337ce0bd8df2178d0e59ac659d231124130b77e038315468cb2bdea",
          "recovery_wire_equivalence": "8cce3da1baf2c842ed7209ffc433c7407d761dd6620b96a0b61e41fc024f4b53",
          "recovery_requests": "59d53b1854e88eb1329c188ad04f6c5622da6fefaea49399905f6f30287bbc58",
          "recovery_raw_files": "49408b0ce33bc9db938f79f9a70f7aabb3e82db4cdca303b556dcd79f77cdb5e",
          "overnight_authority": "fa57a44bcb18a2f78507f0de5070082e79974c72b822509e07df813554f3c0bc",
          "independent_recovery_auditor": "d14d84da09f22a368fea61002d017cd2c632f9e44f406e901509e5729a8345a5",
          "last32_plan": "16ec50aac6fbe71c742f3d107b5fee0841b17288cc053ff6033a73d20cb3f87b",
          "last32_status": "8f69f84a5f87f8420ea49358f68fde1cc2d73bfecd3de4e04fbf3145dd96baf1",
          "last32_preflight": "8e51358195cab18ebb667165bc1fe48b785ae5bfd3909ef3ed1b608d5d18cc63",
          "last32_scorer_preflight": "178dfecb5e0922bcd5639fff65edc3505513a03384e0a99dd9a0fb6a69478e16",
          "last32_wire_equivalence": "b51fc06ccee84f371ec1026ff81b354d56fd5195e2a8580071a2d63e824bdaa3",
          "last32_raw_files": "169dbb9c001459ea09b6619625fdb9e778119f09518b8a59b0e2d0b41c6143ca",
          "multisource_answer_bindings": "c66752ff9e8541135a8cbff6cadeb9f17572b4162bf8469c1d52bed14e137d5d",
          "last32_auditor": "c2a84809ac71542c6f3bed95254fdc3f9b5bb9fc023496934981065285d587c3",
          "last32_authority": "98f439a666bc43b19d721d239d4b20067867068fb5fb9ea88d4660313c5fa6ef",
          "cumulative_request_records": "07ef94f732d107bf53b9b88679124d541a0f0c5c20038e1a0e2c3b131662ebe8"
        }
      }
    }
  ]
}
