{
 "tag": "astra-scry-frozen-full2-20260905",
 "completed_at": "2026-09-05T20:30:27-07:00",
 "solver": "gpt-6-astra only (single-solver; 6 transient ERRORs retried identically across 4 rounds, no cross-model backfill)",
 "graded": 900,
 "macro_f1": 0.8357,
 "f1_sd": 0.3257,
 "f1_se": 0.0109,
 "accuracy_fully_correct": 0.7178,
 "accuracy_95ci": [
  0.6884,
  0.7472
 ],
 "fully_correct": 646,
 "correct_with_excessive": 55,
 "fully_incorrect": 98,
 "accuracy_if_excessive_recovered": 0.7789,
 "comparisons": {
  "prior_hybrid_full900": {
   "macro_f1": 0.7963,
   "accuracy": 0.6418,
   "delta_f1": 0.0394,
   "delta_acc": 0.076
  },
  "kaggle_gemini_dra_published": {
   "f1": 0.819,
   "fully_correct": 0.661,
   "claim_note": "accuracy margin (+5.7pts, SE~1.5pts) is statistically clear; F1 margin (+0.017, SE~0.011) is directionally above but not separable alone"
  },
  "openrouter_25turn_board": {
   "opus5_parallel": 0.77,
   "opus5_perplexity": 0.765,
   "sol_openai_native": 0.75,
   "opus5_exa": 0.705,
   "claim_note": "different harness (excerpts-only, no fetching, proprietary); our 0.7178 sits above their Exa row, below Parallel/Perplexity rows"
  }
 },
 "integrity_disclosures": [
  "Judge = exact published prompt + gemini-2.5-flash via Scry transport, not the organizer SDK; not an official submission",
  "Policy (r14) was developed on dev-50 (shuffle 1-50) and informed by audits of frozen-eval-50 (shuffle 200-250), both subsets of the 900; the remaining 800 were only ever run under frozen policies",
  "The 900 questions were run once before (2026-09-05 hybrid, old policy); no grading outcomes of that run's untuned 800 informed r13/r14 changes",
  "Benchmark-material filtering active at the tool-result boundary",
  "Grading interrupted by wallet exhaustion at 333/900; resumed after a $10 internal promo_credit grant (disclosed; ledger events bench-grading-grant-20260905*)"
 ]
}