{
  "schema_version": 1,
  "board": "JARVI3 DTL Evidence Board",
  "claim_boundary": "Public evidence board; not an official ProgramBench or MathArena leaderboard.",
  "updated": "2026-10-01",
  "entries": [
    {
      "id": "dtl-mathgate-lite-2026-10-01",
      "system": "DTL MathGate lite",
      "track": "public_arithmetic_development_suite",
      "status": "public_seed_evidence",
      "total_cases": 240,
      "exact_answers": 192,
      "expected_refusals": 48,
      "failed": 0,
      "certificate_drift": 0,
      "unit_tests_passed": 18,
      "source_commit": "0884a4f06a65ae73816b1f8c166495aea56eb5d6",
      "report": "evidence/mathgate-2026-10-01.json",
      "rank": null,
      "rank_claim_allowed": false
    },
    {
      "id": "jarvi3-themis-g-programbench-cleanroom-v2",
      "system": "Jarvi3: Themis-G",
      "track": "official_programbench_submission",
      "status": "pending_registry",
      "resolved_instances": 2,
      "benchmark_denominator": 200,
      "recorded_tests": 1037,
      "passed_tests": 1037,
      "failed_tests": 0,
      "archive_integrity": "Archive byte identity verified; generation provenance unresolved",
      "source_repository": "https://github.com/kyal102/jarvi3-themis-g-programbench-2solve",
      "source_commit": "c1240b5175426d9d401a63d3b5a56224ed44d409",
      "registry_pr": "https://github.com/ProgramBench/submissions/pull/27",
      "rank": null,
      "rank_claim_allowed": false,
      "reproduced_date": "2026-10-01",
      "reproduction_report": "evidence/programbench-reproduction-2026-10-01.json",
      "evaluator_version": "1.2.5",
      "raw_tests_passed": 1346,
      "raw_tests_total": 1346,
      "generation_provenance": "unresolved"
    },
    {
      "id": "jarvi3-themis-g-aime-2026-supermath-dtl-public-seed",
      "system": "Jarvi3: Themis-G",
      "track": "case_specific_arithmetic_replay",
      "status": "public_seed_evidence",
      "benchmark": "AIME-2026-style exact-answer set",
      "total_cases": 30,
      "correct": 14,
      "incorrect": 0,
      "abstain": 16,
      "coverage_percent": 46.7,
      "accuracy_overall_percent": 46.7,
      "precision_on_answered_percent": 100.0,
      "generalization_evidence": false,
      "claim_boundary": "Known case identifiers route to problem-specific calculations. This measures public development-case arithmetic replay, not unseen-problem reasoning.",
      "report": "evidence/aime-2026-supermath-dtl-full30.md",
      "rank": null,
      "rank_claim_allowed": false
    }
  ]
}
