{
  "schema_version": "echo-latest-confirmed-evaluation-v1",
  "evaluated_at": "2026-07-15",
  "title": "Latest confirmed evaluation",
  "scope": "Recorded Echo results on matched benchmark questions, with inspectable evidence where available.",
  "candidate_gate": null,
  "benchmarks": [
    {
      "id": "math500",
      "name": "MATH-500",
      "category": "math",
      "cohort": "500 canonical test problems from HuggingFaceH4/MATH-500",
      "evaluated_at": "2026-07-27",
      "echo_correct": 493,
      "echo_total": 500,
      "echo_accuracy": 0.986,
      "reference_name": "Claude Fable",
      "reference_correct": 499,
      "reference_total": 500,
      "reference_accuracy": 0.998,
      "comparison_status": "matched_same_cohort_same_denominator",
      "interpretation": "On the exact same 500 problems and grader, Echo answered 493 correctly and Fable answered 499 correctly. Echo used 63.79% less measured model-inference cost.",
      "echo_rate_card_cost_usd": 4.5771859122,
      "reference_standard_rate_cost_usd": 12.64015,
      "reference_to_echo_cost_ratio": 2.76155486,
      "evidence_status": "Echo · 500 matched questions",
      "cost_status": "Exact accepted output-producing token ledgers repriced at current list rates before markup; infrastructure and failed calls are excluded.",
      "pareto_scope_legend": true,
      "pareto_sources": [
        {
          "label": "Exact 500-question scores and measured model-token ledgers"
        }
      ],
      "pareto_points": [
        {
          "name": "Echo",
          "role": "echo",
          "correct": 493,
          "total": 500,
          "accuracy": 0.986,
          "cost_usd": 4.5771859122,
          "label_dx": 11,
          "label_dy": 24,
          "label_anchor": "start",
          "evidence": "Echo result on the exact 500-question cohort with stored answers, grades, and measured model-token usage."
        },
        {
          "name": "Claude Fable",
          "role": "reference",
          "correct": 499,
          "total": 500,
          "accuracy": 0.998,
          "cost_usd": 12.64015,
          "label_dx": -11,
          "label_dy": -15,
          "label_anchor": "end",
          "evidence": "One exact 500-problem run on the same cohort and grader with an exact token ledger."
        }
      ],
      "pareto_provenance": {
        "release_claim_sha256": "42ee504632c9f5707080f30e8cdcd1be6fa853fb40fd474c64e9e2ccf2efc397",
        "fixture_sha256": "851cd837064af943df95a3c4659cdfaa32ff533122d0e835e8b135f6f3ae0b71",
        "echo_score_artifact_sha256": "20f3c619c4803c98a6ecb663ddd1816e657346b8c5dcfbb76f5670af69ad5b45",
        "echo_rate_card_ledger_sha256": "b6e67d150f007773b2a9bd80cc5bc95027dea027edaede1899d4af56959b7823",
        "echo_repair_provenance_sha256": "b1e297f87ada644701efce7e62f2389aadf51e651a56b4596a6fdb4858cb5a9f",
        "echo_sealed_main_run_sha256": "9ab19e1d5a919f502a2890bb7f41d2fce58c851080101b2e9ff49a8eb0c0420a",
        "echo_repaired_full_run_sha256": "e37253b654c2568ee783aba05b54ec935a4ce912f8348d0b3bedd87cc52c5ddf",
        "echo_sealed_main_rows": 494,
        "echo_reliability_replay_rows": 6,
        "echo_replay_reason": "replace old-image HTTP 503 rows",
        "fable_raw_run_sha256": "c4a2e5562901b994b5171c2a686116773412f3114ce14ec241b6ad5d8a2cd721",
        "grader_commit": "68da5f36c72d83e987bda77155c3bb26898913c0",
        "grader_installed_tree_sha256": "a45a404cf6b418580fdf633e84940d01636216d1f9078dd5b37ed634f301306a",
        "same_exact_frozen_cohort": true
      }
    },
    {
      "id": "livecodebench276",
      "name": "LiveCodeBench",
      "category": "code",
      "cohort": "276 execution-graded tasks",
      "evaluated_at": "2026-07-26",
      "echo_correct": 256,
      "echo_total": 276,
      "echo_accuracy": 0.9275362319,
      "reference_name": "Claude Fable",
      "reference_correct": 254,
      "reference_total": 276,
      "reference_accuracy": 0.9202898551,
      "comparison_status": "matched_same_cohort_same_denominator",
      "interpretation": "On the same 276 execution-graded tasks, Echo solved 256 and Fable solved 254. Echo leads by 2 tasks (0.72 percentage points); this point estimate is not evidence of statistically significant superiority.",
      "echo_rate_card_cost_usd": 5.354451624,
      "reference_standard_rate_cost_usd": 8.70862,
      "reference_to_echo_cost_ratio": 1.6264261238,
      "evidence_status": "Echo · 276 matched tasks",
      "cost_status": "Exact successful generative model token ledger repriced at current OpenRouter public list rates before markup; safety, sandbox and infrastructure costs excluded.",
      "pareto_scope_legend": true,
      "pareto_sources": [
        {
          "label": "Echo exact 276-task score and measured model-token ledger"
        },
        {
          "label": "Nemotron-3 Ultra technical report · LiveCodeBench v6 comparison panel",
          "url": "https://research.nvidia.com/labs/nemotron/files/NVIDIA-Nemotron-3-Ultra-Technical-Report.pdf"
        }
      ],
      "pareto_points": [
        {
          "name": "Echo",
          "role": "echo",
          "correct": 256,
          "total": 276,
          "accuracy": 0.9275362319,
          "cost_usd": 5.354451624,
          "label_dx": -11,
          "label_dy": -15,
          "label_anchor": "end",
          "evidence": "Echo result on the exact same 276 tasks and official evaluator."
        },
        {
          "name": "Kimi K2.6",
          "role": "openweight",
          "correct": 254,
          "total": 276,
          "accuracy": 0.9202898551,
          "cost_usd": 6.555729448,
          "label_dx": 11,
          "label_dy": 22,
          "label_anchor": "start",
          "evidence": "Direct single-model control on the same 276 tasks; official scorer and recorded token ledger."
        },
        {
          "name": "Claude Fable",
          "role": "reference",
          "correct": 254,
          "total": 276,
          "accuracy": 0.9202898551,
          "cost_usd": 8.70862,
          "label_dx": -11,
          "label_dy": -15,
          "label_anchor": "end",
          "evidence": "Exact same 276 tasks and official evaluator."
        }
      ],
      "score_context": [
        {"name": "DeepSeek V4 Pro · published v6", "score": 92.5, "role": "external", "scope": "provider-reported LiveCodeBench v6; different cohort", "source": "NVIDIA Nemotron 3 Ultra technical report"},
        {"name": "DeepSeek V4 Flash · published v6", "score": 90.9, "role": "external", "scope": "provider-reported LiveCodeBench v6; different cohort", "source": "NVIDIA Nemotron 3 Ultra technical report"},
        {"name": "Kimi K2.6 · published v6", "score": 90.2, "role": "external", "scope": "provider-reported LiveCodeBench v6; different cohort", "source": "NVIDIA Nemotron 3 Ultra technical report"},
        {"name": "Nemotron 3 Ultra · published v6", "score": 89.0, "role": "external", "scope": "provider-reported LiveCodeBench v6; different cohort", "source": "NVIDIA Nemotron 3 Ultra technical report"},
        {"name": "Qwen3.5-397B · published v6", "score": 79.3, "role": "external", "scope": "provider-reported LiveCodeBench v6; different cohort", "source": "NVIDIA Nemotron 3 Ultra technical report"}
      ],
      "pareto_provenance": {
        "echo_component_cost_ledger_sha256": "768b1ee4752b332ef5de300d0c47d8a19378ad95c882454d598f735f064118fd",
        "kimi_candidate_sha256": "ab3990706c1b0de9ef0b25b93187f759956d3f8cc939afdbc58e2eb6b2eb5731",
        "kimi_official_score_sha256": "ace47cce8545776f7a7d995953df24519b1259a2c6bc7ba70425b71c1f292f31",
        "openrouter_pricing_catalog_sha256": "4808dc30d333505b363709cfb8b6dd71113928f147ef6172bfbb0e3d80fb7036",
        "same_ids_prompts_dataset_and_evaluator": true
      },
      "evaluation_provenance": {
        "fixture_rows": 276,
        "fixture_sha256": "28edd7afd1a0386e69510f23da52db5b0deafc80ad88498319508f3126e2bbd9",
        "dataset_arrow_sha256": "9e162b1776afaf85da80cf505839dd39f861590e9ba6ffa63c72fba1dc0146c6",
        "official_evaluator_repository": "LiveCodeBench",
        "official_evaluator_commit": "28fef95ea8c9f7a547c8329f2cd3d32b92c1fa24",
        "echo_score_sha256": "8044e55d011a31dfa24dfae427efc207d2af2df0d20a571ae12fb7346aaf4667",
        "reference_score_sha256": "1afb98724b3031ff52e553be1250153d112fc1a0e722011a29a7f84f3917bd0d",
        "reference_candidate_sha256": "3b57f2393dd33bfa85a257debebf48b7dcc99efeeef5364d671b68e734257002",
        "same_ids_prompts_dataset_and_evaluator": true
      }
    },
    {
      "id": "mmlupro140",
      "name": "MMLU-Pro frozen cohort",
      "category": "reasoning",
      "cohort": "140 frozen questions",
      "evaluated_at": "2026-07-12",
      "echo_correct": 123,
      "echo_total": 140,
      "echo_accuracy": 0.8785714286,
      "reference_name": "Claude Fable",
      "reference_correct": 126,
      "reference_total": 140,
      "reference_accuracy": 0.9,
      "comparison_status": "matched_same_cohort_same_denominator",
      "interpretation": "Echo trails Fable by 3 questions (2.14 percentage points) on the same frozen subset.",
      "presentation": "pareto",
      "pareto_scope_legend": true,
      "pareto_points": [
        {
          "name": "Echo",
          "role": "echo",
          "accuracy": 0.8785714286,
          "cost_usd": 1.42897291,
          "label_dx": 12,
          "label_dy": -36,
          "label_anchor": "start",
          "evidence": "Exact 123/140 score; exact accepted-call token ledger repriced at current OpenRouter list rates. The run used the Kimi K2.6 floor, but this point is Echo's full exact-cohort result."
        }
      ],
      "score_context": [
        {"name": "Claude Fable", "score": 90.0, "role": "reference", "scope": "exact same frozen 140-question subset; measured token ledger unavailable", "source": "Local exact-subset score ledger"},
        {"name": "Gemini 3.1 Pro", "score": 91.16, "role": "external", "scope": "published full 12,032-question MMLU-Pro; different cohort", "source": "Published MMLU-Pro leaderboard"},
        {"name": "Claude 4.6 Opus", "score": 89.1, "role": "external", "scope": "published full 12,032-question MMLU-Pro; different cohort", "source": "Published MMLU-Pro leaderboard"},
        {"name": "Qwen3.5 397B A17B", "score": 88.3, "role": "external", "scope": "provider-reported full MMLU-Pro panel; different cohort", "source": "NVIDIA Nemotron 3 Ultra technical report"},
        {"name": "Kimi K2.6", "score": 88.1, "role": "external", "scope": "provider-reported full MMLU-Pro panel; different cohort", "source": "NVIDIA Nemotron 3 Ultra technical report"}
      ],
      "pareto_sources": [
        {
          "label": "Exact 140-row score and Echo token ledger (local frozen evidence)"
        },
        {
          "label": "Published MMLU-Pro leaderboard",
          "url": "https://huggingface.co/datasets/TIGER-Lab/mmlu_pro_leaderboard_submission/resolve/main/results.csv"
        },
        {
          "label": "Nemotron-3 Ultra technical report",
          "url": "https://research.nvidia.com/labs/nemotron/files/NVIDIA-Nemotron-3-Ultra-Technical-Report.pdf"
        }
      ],
      "echo_rate_card_cost_usd": 1.42897291,
      "reference_standard_rate_cost_usd": null,
      "cost_status": "Echo's exact 139-call output-producing token ledger repriced at current OpenRouter list rates is $1.429. One missing-canonical row produced no tokens and is counted wrong in the 123/140 score. Fable and published models have no comparable measured token ledger, so they receive no cost-axis position."
    },
    {
      "id": "gpqa100dev",
      "name": "GPQA Diamond development cohort",
      "category": "science",
      "cohort": "100 frozen development questions",
      "evaluated_at": "2026-07-13",
      "echo_correct": 94,
      "echo_total": 100,
      "echo_accuracy": 0.94,
      "reference_name": "Claude Fable",
      "reference_correct": 94,
      "reference_total": 100,
      "reference_accuracy": 0.94,
      "comparison_status": "matched_development_evidence",
      "interpretation": "Echo matches Fable on this exact cohort. The cohort was used during policy development, so this is development evidence—not an independent holdout.",
      "echo_rate_card_cost_usd": 4.572658285,
      "reference_standard_rate_cost_usd": 6.72623,
      "cost_status": "Same 100-question development cohort; measured successful model-inference cost. This remains development evidence, not an independent holdout.",
      "pareto_scope_legend": true,
      "pareto_sources": [
        {
          "label": "Echo development ledger and exact 100-row comparison (local frozen evidence)"
        },
        {
          "label": "OpenAI GPT-5.6 comparison panel",
          "url": "https://openai.com/index/gpt-5-6/"
        },
        {
          "label": "Gemini 3.1 Pro model card",
          "url": "https://deepmind.google/models/model-cards/gemini-3-1-pro"
        },
        {
          "label": "Nemotron-3 Ultra technical report",
          "url": "https://research.nvidia.com/labs/nemotron/files/NVIDIA-Nemotron-3-Ultra-Technical-Report.pdf"
        }
      ],
      "pareto_points": [
        {
          "name": "Echo",
          "role": "echo",
          "correct": 94,
          "total": 100,
          "accuracy": 0.94,
          "cost_usd": 4.572658285,
          "label_dx": -11,
          "label_dy": -15,
          "label_anchor": "end",
          "evidence": "Measured on the exact 100-question development cohort."
        },
        {
          "name": "Kimi K2.6",
          "role": "openweight",
          "correct": 91,
          "total": 100,
          "accuracy": 0.91,
          "cost_usd": 5.26541612,
          "label_dx": 11,
          "label_dy": 22,
          "label_anchor": "start",
          "evidence": "Direct single-model control on the same 100 questions; recorded tokens repriced at current OpenRouter list rate."
        },
        {
          "name": "Claude Fable",
          "role": "reference",
          "correct": 94,
          "total": 100,
          "accuracy": 0.94,
          "cost_usd": 6.72623,
          "label_dx": 11,
          "label_dy": -15,
          "label_anchor": "start",
          "evidence": "Exact same 100-question development cohort."
        }
      ],
      "score_context": [
        {"name": "GPT-5.6 Sol", "score": 94.6, "role": "external", "scope": "published full 198-question GPQA Diamond; different cohort", "source": "OpenAI GPT-5.6 launch report"},
        {"name": "Claude Mythos 5", "score": 94.1, "role": "external", "scope": "published full 198-question GPQA Diamond; different cohort", "source": "OpenAI GPT-5.6 comparison panel"},
        {"name": "Gemini 3.1 Pro", "score": 94.3, "role": "external", "scope": "published full 198-question GPQA Diamond; different cohort", "source": "Gemini 3.1 Pro model card"},
        {"name": "Claude Fable 5", "score": 92.6, "role": "external", "scope": "published full 198-question GPQA Diamond; different cohort", "source": "OpenAI GPT-5.6 comparison panel"},
        {"name": "Claude 4.6 Opus", "score": 91.3, "role": "external", "scope": "published full 198-question GPQA Diamond; different cohort", "source": "Gemini 3.1 Pro model card"},
        {"name": "Kimi K2.6 · published", "score": 91.0, "role": "external", "scope": "provider-reported full GPQA Diamond panel; different cohort", "source": "NVIDIA Nemotron 3 Ultra technical report"},
        {"name": "DeepSeek V4 Flash", "score": 88.5, "role": "external", "scope": "provider-reported full GPQA Diamond panel; different cohort", "source": "NVIDIA Nemotron 3 Ultra technical report"},
        {"name": "Qwen3.5-397B", "score": 87.1, "role": "external", "scope": "provider-reported full GPQA Diamond panel; different cohort", "source": "NVIDIA Nemotron 3 Ultra technical report"}
      ],
      "pareto_provenance": {
        "echo_fable_summary_sha256": "dfec68df305e942b8c912a779524b096f122ffa4b1ae2e5153c45d9e02d42eae",
        "materialization_report_sha256": "8b1552e00b8ce40f6b86fb1ea8c0110b7be74858aa6f4ca72759a724c9f46412",
        "kimi_candidate_sha256": "31d959e49c2de3f25cc2f9c5a4867335cd8eb92aa7f4efb2126b38599bd66a9c",
        "openrouter_pricing_catalog_sha256": "4808dc30d333505b363709cfb8b6dd71113928f147ef6172bfbb0e3d80fb7036"
      }
    }
  ],
  "row_inspection_status": "aggregate_only",
  "row_inspection_note": "The score summaries and cost ledgers hash-match, but the referenced current answer ledgers are absent from this checkout. These two summaries are therefore not linked to the historical row browser below.",
  "sources": [
    {
      "role": "MATH-500 exact release claim",
      "sha256": "42ee504632c9f5707080f30e8cdcd1be6fa853fb40fd474c64e9e2ccf2efc397"
    },
    {
      "role": "MATH-500 Echo score artifact",
      "sha256": "20f3c619c4803c98a6ecb663ddd1816e657346b8c5dcfbb76f5670af69ad5b45"
    },
    {
      "role": "MATH-500 Echo rate-card ledger",
      "sha256": "b6e67d150f007773b2a9bd80cc5bc95027dea027edaede1899d4af56959b7823"
    },
    {
      "role": "MATH-500 Echo repair provenance",
      "sha256": "b1e297f87ada644701efce7e62f2389aadf51e651a56b4596a6fdb4858cb5a9f"
    },
    {
      "role": "MATH-500 Echo sealed main run",
      "sha256": "9ab19e1d5a919f502a2890bb7f41d2fce58c851080101b2e9ff49a8eb0c0420a"
    },
    {
      "role": "MATH-500 Echo repaired full run",
      "sha256": "e37253b654c2568ee783aba05b54ec935a4ce912f8348d0b3bedd87cc52c5ddf"
    },
    {
      "role": "MATH-500 Fable raw run",
      "sha256": "c4a2e5562901b994b5171c2a686116773412f3114ce14ec241b6ad5d8a2cd721"
    },
    {
      "role": "LiveCodeBench score summary",
      "sha256": "8044e55d011a31dfa24dfae427efc207d2af2df0d20a571ae12fb7346aaf4667"
    },
    {
      "role": "LiveCodeBench matched Fable score summary",
      "sha256": "1afb98724b3031ff52e553be1250153d112fc1a0e722011a29a7f84f3917bd0d"
    },
    {
      "role": "LiveCodeBench matched Fable candidate ledger",
      "sha256": "3b57f2393dd33bfa85a257debebf48b7dcc99efeeef5364d671b68e734257002"
    },
    {
      "role": "Kimi K2.6 direct LiveCodeBench control",
      "sha256": "ab3990706c1b0de9ef0b25b93187f759956d3f8cc939afdbc58e2eb6b2eb5731"
    },
    {
      "role": "OpenRouter pricing catalog snapshot",
      "sha256": "4808dc30d333505b363709cfb8b6dd71113928f147ef6172bfbb0e3d80fb7036"
    }
  ]
}
