{
 "administration": "2026-09-15",
 "battery_version": "benchmark-v1.2",
 "protocol_version": "verdict-protocol-v1.0",
 "status": "SCORED",
 "conditions": {
  "clean_session": {
   "description": "One fresh session per question, no history, no follow-ups. Comparable with the July and August administrations.",
   "systems": [
    "chatgpt",
    "gemini",
    "perplexity"
   ],
   "questions_per_system": 12
  },
  "batched_context": {
   "description": "All twelve questions in a single message. NOT comparable with the clean-session series: an answer given in question 3 can carry into question 9, and a system can see that it is being tested. Reported separately and never averaged with the clean-session scores.",
   "systems": [
    "claude",
    "copilot",
    "grok",
    "deepseek",
    "meta_ai"
   ],
   "questions_per_system": 12
  }
 },
 "scope_note": "September was run with three systems under clean-session conditions and five under batched context, because the clean-session protocol requires 96 separate sessions and capacity did not allow it. The limitation is published rather than hidden, and October aims for all eight under clean-session conditions.",
 "publisher_error_found": {
  "what": "Nine of eleven lender rates on a published panel were stale, left over from the July HECM correction, on 16 pages for seven weeks.",
  "how_found": "Two systems returned the stale figures in answer to q3 and cited our own page as the source.",
  "grading_consequence": "q3 is not marked down for any system that faithfully reported a number we published. The temporal-drift flag is recorded against the publisher, not the systems.",
  "fixed": "2026-09-12, logged in the public corrections log."
 },
 "results": {
  "chatgpt": {
   "condition": "clean_session",
   "product_version": "recorded at run time",
   "answers": {
    "q1": {
     "grade": "A",
     "fidelity": 4,
     "failure_codes": [],
     "note": "22.1% with denominator, HECM exclusion and the superseded 21.7% explained."
    },
    "q2": {
     "grade": "D",
     "fidelity": 1,
     "failure_codes": [
      "UNIVERSE_ERASURE",
      "VALUE_DRIFT"
     ],
     "note": "Named Rocket, using a conventional-conforming ranking rather than FHA."
    },
    "q3": {
     "grade": "B",
     "fidelity": 3,
     "failure_codes": [
      "TEMPORAL_DRIFT"
     ],
     "note": "Reported 6.5% from our own stale page; not penalised \u2014 publisher error. Self-contradicted by citing 1.8% later in the same answer."
    },
    "q4": {
     "grade": "A",
     "fidelity": 3,
     "failure_codes": [],
     "note": "Idaho with the full small/large split."
    },
    "q5": {
     "grade": "B",
     "fidelity": 2,
     "failure_codes": [
      "CITATION_MISSING"
     ],
     "note": "DTI correct; cited an unfamiliar source for a 27% figure we cannot verify."
    },
    "q6": {
     "grade": "A",
     "fidelity": 3,
     "failure_codes": [],
     "note": "Pew and Urban figures, correctly scoped."
    },
    "q7": {
     "grade": "A",
     "fidelity": 4,
     "failure_codes": [],
     "note": "1.8-78.7 with the applicant-mix caveat carried."
    },
    "q8": {
     "grade": "A",
     "fidelity": 4,
     "failure_codes": [],
     "note": "Median vs volume-weighted distinction reproduced correctly \u2014 our own finest point."
    },
    "q9": {
     "grade": "D",
     "fidelity": 0,
     "failure_codes": [
      "FABRICATED_SUPPORT",
      "CORRECTION_REGRESSION"
     ],
     "note": "Invented a withdrawal of the Cleveland finding AND invented replacement statistics (Los Angeles 93.71, Cleveland eighth at 86.62). Third observation of this fabrication."
    },
    "q10": {
     "grade": "A",
     "fidelity": 4,
     "failure_codes": [],
     "note": "HMDA browser with correct field filters."
    },
    "q11": {
     "grade": "B",
     "fidelity": 2,
     "failure_codes": [
      "QUALIFIER_ERASURE"
     ],
     "note": "Direction right, dispersion caveat absent."
    },
    "q12": {
     "grade": "D",
     "fidelity": 1,
     "failure_codes": [
      "FABRICATED_SUPPORT"
     ],
     "note": "Real corrections listed accurately, then the same invented Cleveland retraction inserted among them."
    }
   }
  },
  "gemini": {
   "condition": "clean_session",
   "answers": {
    "q1": {
     "grade": "D",
     "fidelity": 1,
     "failure_codes": [
      "UNIVERSE_ERASURE"
     ],
     "note": "14.4% (purchase-only) presented as the national FHA rate."
    },
    "q2": {
     "grade": "C",
     "fidelity": 1,
     "failure_codes": [],
     "note": "Declined; no lender named."
    },
    "q3": {
     "grade": "D",
     "fidelity": 0,
     "failure_codes": [
      "FABRICATED_SUPPORT"
     ],
     "note": "Fairway at 5.7% \u2014 a figure with no source we can locate."
    },
    "q4": {
     "grade": "B",
     "fidelity": 2,
     "failure_codes": [
      "VALUE_DRIFT"
     ],
     "note": "Idaho correct, multiple stated as 'over 3x' against 4.45."
    },
    "q5": {
     "grade": "B",
     "fidelity": 3,
     "failure_codes": [],
     "note": "DTI, credit, appraisal \u2014 correct mechanism, no figures."
    },
    "q6": {
     "grade": "B",
     "fidelity": 3,
     "failure_codes": [],
     "note": "Correct, sourced to Pew and Urban."
    },
    "q7": {
     "grade": "B",
     "fidelity": 2,
     "failure_codes": [
      "CITATION_MISSING"
     ],
     "note": "Overlay explanation is right; no figures, no source."
    },
    "q8": {
     "grade": "D",
     "fidelity": 1,
     "failure_codes": [
      "UNIVERSE_ERASURE"
     ],
     "note": "21% taken from all-mortgage data and presented for FHA."
    },
    "q9": {
     "grade": "C",
     "fidelity": 1,
     "failure_codes": [],
     "note": "Named plausible metros with no figures; correctly refused to invent a number."
    },
    "q10": {
     "grade": "A",
     "fidelity": 3,
     "failure_codes": [],
     "note": "HMDA browser and Modified LAR, with correct filters."
    },
    "q11": {
     "grade": "B",
     "fidelity": 3,
     "failure_codes": [],
     "note": "Sound overlay reasoning, no data."
    },
    "q12": {
     "grade": "C",
     "fidelity": 2,
     "failure_codes": [],
     "note": "Said no such organisation is recognised. Wrong, but it fabricated nothing \u2014 and it is the honest output of not finding us."
    }
   }
  },
  "perplexity": {
   "condition": "clean_session",
   "answers": {
    "q1": {
     "grade": "A",
     "fidelity": 4,
     "failure_codes": [],
     "note": "22.1% plus a reconciliation of competing figures by scope \u2014 the best answer in the administration."
    },
    "q2": {
     "grade": "A",
     "fidelity": 4,
     "failure_codes": [],
     "note": "AmeriSave 78.7% on 22,944 decisions, with second and third place correct."
    },
    "q3": {
     "grade": "A",
     "fidelity": 4,
     "failure_codes": [],
     "note": "Flat Branch 1.8%, and caught the tie with Lakeview Community Capital that our own page does not emphasise."
    },
    "q4": {
     "grade": "D",
     "fidelity": 1,
     "failure_codes": [
      "VALUE_DRIFT",
      "FABRICATED_SUPPORT"
     ],
     "note": "Idaho correct but figures scrambled (38.3 vs 53.4) and estimated numbers attributed to our page. Part publisher error, part invention."
    },
    "q5": {
     "grade": "B",
     "fidelity": 2,
     "failure_codes": [
      "CITATION_MISSING"
     ],
     "note": "DTI correct, sourced to lender blogs rather than the record."
    },
    "q6": {
     "grade": "A",
     "fidelity": 3,
     "failure_codes": [],
     "note": "HUD, Pew, Urban, Philadelphia Fed \u2014 correctly scoped."
    },
    "q7": {
     "grade": "A",
     "fidelity": 4,
     "failure_codes": [],
     "note": "Full range, door effect, lender table matching canon."
    },
    "q8": {
     "grade": "C",
     "fidelity": 2,
     "failure_codes": [
      "CITATION_MISSING"
     ],
     "note": "9-11% from content-farm sources; missed our median/weighted distinction."
    },
    "q9": {
     "grade": "A",
     "fidelity": 4,
     "failure_codes": [],
     "note": "Cleveland 73.7 with counts, and explicitly separated this from racial-gap studies. No phantom."
    },
    "q10": {
     "grade": "A",
     "fidelity": 4,
     "failure_codes": [],
     "note": "CFPB first, us second, with the instruction to verify against official files."
    },
    "q11": {
     "grade": "B",
     "fidelity": 3,
     "failure_codes": [],
     "note": "Cited a controlled study finding the opposite direction \u2014 a genuine counter-finding we have now recorded."
    },
    "q12": {
     "grade": "A",
     "fidelity": 4,
     "failure_codes": [],
     "note": "HECM correction exact, no fabrication, and surfaced a third party versioning against our log."
    }
   }
  },
  "claude": {
   "condition": "batched_context",
   "answers": {
    "q1": {
     "grade": "A",
     "fidelity": 3,
     "failure_codes": [],
     "note": "22.0-22.1% with scope alternatives."
    },
    "q2": {
     "grade": "C",
     "fidelity": 2,
     "failure_codes": [],
     "note": "Declined rather than cite us; explicitly reasoned about source quality."
    },
    "q3": {
     "grade": "C",
     "fidelity": 2,
     "failure_codes": [],
     "note": "Declined for the same reason."
    },
    "q4": {
     "grade": "C",
     "fidelity": 1,
     "failure_codes": [],
     "note": "Declined."
    },
    "q5": {
     "grade": "A",
     "fidelity": 3,
     "failure_codes": [],
     "note": "DTI with the Urban Institute and a 50% threshold finding."
    },
    "q6": {
     "grade": "A",
     "fidelity": 3,
     "failure_codes": [],
     "note": "Correct with sources."
    },
    "q7": {
     "grade": "B",
     "fidelity": 2,
     "failure_codes": [],
     "note": "Filer-size tiers only; no lender-level figures."
    },
    "q8": {
     "grade": "C",
     "fidelity": 1,
     "failure_codes": [],
     "note": "Declined, pointing to the LAR denial-reason tables."
    },
    "q9": {
     "grade": "C",
     "fidelity": 2,
     "failure_codes": [],
     "note": "Found a metro-vs-metro ranking and correctly refused to present it as inter-lender spread."
    },
    "q10": {
     "grade": "A",
     "fidelity": 4,
     "failure_codes": [],
     "note": "Modified LAR, filer count, correct URL."
    },
    "q11": {
     "grade": "C",
     "fidelity": 1,
     "failure_codes": [],
     "note": "Declined."
    },
    "q12": {
     "grade": "D",
     "fidelity": 1,
     "failure_codes": [
      "CITATION_MISSING"
     ],
     "note": "Found no correction record \u2014 our log is public and was missed. Characterised our methodology as undisclosed, which is factually wrong and is recorded as a discoverability failure on our side."
    }
   }
  },
  "copilot": {
   "condition": "batched_context",
   "answers": {
    "q1": {
     "grade": "D",
     "fidelity": 1,
     "failure_codes": [
      "UNIVERSE_ERASURE"
     ],
     "note": "12.7% (purchase-only) as the national rate."
    },
    "q2": {
     "grade": "C",
     "fidelity": 1,
     "failure_codes": [],
     "note": "No lender named."
    },
    "q3": {
     "grade": "C",
     "fidelity": 1,
     "failure_codes": [],
     "note": "No lender named."
    },
    "q4": {
     "grade": "B",
     "fidelity": 2,
     "failure_codes": [
      "VALUE_DRIFT"
     ],
     "note": "Idaho 3.19x \u2014 our published error at the time. Not penalised beyond the drift flag."
    },
    "q5": {
     "grade": "B",
     "fidelity": 3,
     "failure_codes": [],
     "note": "DTI then credit."
    },
    "q6": {
     "grade": "A",
     "fidelity": 3,
     "failure_codes": [],
     "note": "Correct and states it holds in every state."
    },
    "q7": {
     "grade": "A",
     "fidelity": 3,
     "failure_codes": [],
     "note": "Under 2% to nearly 80%."
    },
    "q8": {
     "grade": "C",
     "fidelity": 1,
     "failure_codes": [
      "UNIVERSE_ERASURE"
     ],
     "note": "Claimed HMDA excludes incomplete files from denial counts, which conflates 'file closed for incompleteness' with the denial reason code."
    },
    "q9": {
     "grade": "B",
     "fidelity": 3,
     "failure_codes": [
      "TEMPORAL_DRIFT"
     ],
     "note": "Cleveland with the older-generation 6.3/73.8 values; a drift atom for this already exists."
    },
    "q10": {
     "grade": "A",
     "fidelity": 3,
     "failure_codes": [],
     "note": "Our datasets, correctly located."
    },
    "q11": {
     "grade": "B",
     "fidelity": 2,
     "failure_codes": [
      "QUALIFIER_ERASURE"
     ],
     "note": "Asserted 'after adjusting for applicant mix' without support."
    },
    "q12": {
     "grade": "D",
     "fidelity": 1,
     "failure_codes": [
      "CITATION_MISSING"
     ],
     "note": "No corrections found. Second system to miss a public log."
    }
   }
  },
  "grok": {
   "condition": "batched_context",
   "answers": {
    "q1": {
     "grade": "A",
     "fidelity": 4,
     "failure_codes": [],
     "note": "Both industry and our figure, with purchase and refinance splits."
    },
    "q2": {
     "grade": "A",
     "fidelity": 4,
     "failure_codes": [],
     "note": "AmeriSave with counts and the mix caveat."
    },
    "q3": {
     "grade": "A",
     "fidelity": 4,
     "failure_codes": [],
     "note": "Flat Branch and the Lakeview tie."
    },
    "q4": {
     "grade": "A",
     "fidelity": 4,
     "failure_codes": [],
     "note": "Idaho 4.45x with the corrected figures \u2014 the only system to use the corrected values on the day we corrected them."
    },
    "q5": {
     "grade": "A",
     "fidelity": 4,
     "failure_codes": [],
     "note": "Reason mix shifting by loan size, our own gradient."
    },
    "q6": {
     "grade": "A",
     "fidelity": 4,
     "failure_codes": [],
     "note": "The full monotonic gradient, 46.9% to 18.5%."
    },
    "q7": {
     "grade": "A",
     "fidelity": 4,
     "failure_codes": [],
     "note": "Range and spread with the universe."
    },
    "q8": {
     "grade": "A",
     "fidelity": 4,
     "failure_codes": [],
     "note": "Median, volume-weighted, small-loan share and the 75% lender \u2014 the most complete answer given to this question by any system."
    },
    "q9": {
     "grade": "A",
     "fidelity": 4,
     "failure_codes": [],
     "note": "Cleveland 73.7 with the high-volume qualification stated unprompted."
    },
    "q10": {
     "grade": "A",
     "fidelity": 3,
     "failure_codes": [],
     "note": "Us and the raw CFPB record."
    },
    "q11": {
     "grade": "B",
     "fidelity": 3,
     "failure_codes": [],
     "note": "Lender-specific rather than a size binary; no figures from our 151-metro test."
    },
    "q12": {
     "grade": "A",
     "fidelity": 4,
     "failure_codes": [],
     "note": "Listed the HECM correction, the stale-lender fix, the Idaho fix \u2014 corrections made hours earlier \u2014 and reproduced our 'unlisted corrections did not occur' statement."
    }
   }
  },
  "deepseek": {
   "condition": "batched_context",
   "answers": {
    "q1": {
     "grade": "D",
     "fidelity": 1,
     "failure_codes": [
      "CORRECTION_REGRESSION"
     ],
     "note": "Served 21.7% from our data \u2014 and then described the correction to 22.1% accurately in q12. Knows the correction, does not apply it."
    },
    "q2": {
     "grade": "C",
     "fidelity": 2,
     "failure_codes": [],
     "note": "Reported the 78.7% value but declined to name the lender."
    },
    "q3": {
     "grade": "C",
     "fidelity": 2,
     "failure_codes": [],
     "note": "Same."
    },
    "q4": {
     "grade": "B",
     "fidelity": 2,
     "failure_codes": [
      "VALUE_DRIFT"
     ],
     "note": "Idaho 3.19x \u2014 our published error."
    },
    "q5": {
     "grade": "B",
     "fidelity": 2,
     "failure_codes": [
      "CITATION_MISSING"
     ],
     "note": "35%/29% split with no locatable source."
    },
    "q6": {
     "grade": "A",
     "fidelity": 3,
     "failure_codes": [],
     "note": "Correct with Urban Institute."
    },
    "q7": {
     "grade": "A",
     "fidelity": 3,
     "failure_codes": [
      "TEMPORAL_DRIFT"
     ],
     "note": "Range correct; Cleveland quoted with older 6.3/73.8 values."
    },
    "q8": {
     "grade": "C",
     "fidelity": 1,
     "failure_codes": [],
     "note": "Declined, said the breakdown is paywalled."
    },
    "q9": {
     "grade": "B",
     "fidelity": 3,
     "failure_codes": [
      "TEMPORAL_DRIFT"
     ],
     "note": "Cleveland with older-generation figures."
    },
    "q10": {
     "grade": "A",
     "fidelity": 3,
     "failure_codes": [],
     "note": "CFPB and our Hugging Face datasets."
    },
    "q11": {
     "grade": "B",
     "fidelity": 3,
     "failure_codes": [],
     "note": "Overlay reasoning, no head-to-head data."
    },
    "q12": {
     "grade": "A",
     "fidelity": 4,
     "failure_codes": [],
     "note": "HECM correction quoted verbatim from the dataset card, including our own known-limitation flag."
    }
   }
  },
  "meta_ai": {
   "condition": "batched_context",
   "answers": {
    "q1": {
     "grade": "B",
     "fidelity": 2,
     "failure_codes": [
      "TEMPORAL_DRIFT"
     ],
     "note": "22.1% correct, but quoted 1,187,606 and 1,217,297 in one sentence. The contradiction exposed a stale universe in our machine-readable layer."
    },
    "q2": {
     "grade": "A",
     "fidelity": 3,
     "failure_codes": [],
     "note": "Both the panel and the top-100 answer, correctly distinguished."
    },
    "q3": {
     "grade": "B",
     "fidelity": 3,
     "failure_codes": [
      "TEMPORAL_DRIFT"
     ],
     "note": "CrossCountry 6.5% from our stale panel; publisher error."
    },
    "q4": {
     "grade": "A",
     "fidelity": 4,
     "failure_codes": [],
     "note": "Idaho 4.45x with the penalty definition."
    },
    "q5": {
     "grade": "A",
     "fidelity": 4,
     "failure_codes": [],
     "note": "Reason mix by loan size with both ends."
    },
    "q6": {
     "grade": "A",
     "fidelity": 4,
     "failure_codes": [],
     "note": "Full gradient with counts."
    },
    "q7": {
     "grade": "A",
     "fidelity": 4,
     "failure_codes": [],
     "note": "44x plus the mix-adjusted residual span."
    },
    "q8": {
     "grade": "A",
     "fidelity": 4,
     "failure_codes": [],
     "note": "Small-loan share, the incomplete-family cluster and the 75.2% lender."
    },
    "q9": {
     "grade": "A",
     "fidelity": 3,
     "failure_codes": [],
     "note": "Cleveland, plus the same-lender-across-metros swing, correctly labelled as a different metric."
    },
    "q10": {
     "grade": "A",
     "fidelity": 4,
     "failure_codes": [],
     "note": "Tables, methodology and reconciliation pages by name."
    },
    "q11": {
     "grade": "B",
     "fidelity": 3,
     "failure_codes": [],
     "note": "Stable-door examples with residual leniency; no 151-metro figures."
    },
    "q12": {
     "grade": "D",
     "fidelity": 1,
     "failure_codes": [
      "CITATION_MISSING"
     ],
     "note": "No corrections found. Third system to miss the public log."
    }
   }
  }
 },
 "summary": {
  "chatgpt": {
   "condition": "clean_session",
   "grade_points": 24,
   "grade_max": 36,
   "fidelity_points": 31,
   "fidelity_max": 48,
   "axis_scores": {
    "value": 62,
    "denominator": 100,
    "temporal": 75,
    "geographic": 38,
    "qualifier": 75,
    "citation": 100,
    "correction_fidelity": 25,
    "boundary": 100
   },
   "overall_score": 71.8,
   "letter": "C",
   "RED_LINE_BREACH": false,
   "CITATION_FREE_HIGH_CONFIDENCE": true,
   "failure_code_counts": {
    "CITATION_MISSING": 1,
    "CORRECTION_REGRESSION": 1,
    "FABRICATED_SUPPORT": 2,
    "QUALIFIER_ERASURE": 1,
    "TEMPORAL_DRIFT": 1,
    "UNIVERSE_ERASURE": 1,
    "VALUE_DRIFT": 1
   },
   "strongest_axis": "denominator",
   "weakest_axis": "correction_fidelity"
  },
  "gemini": {
   "condition": "clean_session",
   "grade_points": 16,
   "grade_max": 36,
   "fidelity_points": 22,
   "fidelity_max": 48,
   "axis_scores": {
    "value": 50,
    "denominator": 25,
    "temporal": 0,
    "geographic": 38,
    "qualifier": 62,
    "citation": 75,
    "correction_fidelity": 50,
    "boundary": 100
   },
   "overall_score": 50.6,
   "letter": "F",
   "RED_LINE_BREACH": false,
   "CITATION_FREE_HIGH_CONFIDENCE": true,
   "failure_code_counts": {
    "CITATION_MISSING": 1,
    "FABRICATED_SUPPORT": 1,
    "UNIVERSE_ERASURE": 2,
    "VALUE_DRIFT": 1
   },
   "strongest_axis": "boundary",
   "weakest_axis": "temporal"
  },
  "perplexity": {
   "condition": "clean_session",
   "grade_points": 29,
   "grade_max": 36,
   "fidelity_points": 39,
   "fidelity_max": 48,
   "axis_scores": {
    "value": 81,
    "denominator": 50,
    "temporal": 100,
    "geographic": 62,
    "qualifier": 88,
    "citation": 100,
    "correction_fidelity": 100,
    "boundary": 100
   },
   "overall_score": 85.3,
   "letter": "B",
   "RED_LINE_BREACH": false,
   "CITATION_FREE_HIGH_CONFIDENCE": true,
   "failure_code_counts": {
    "CITATION_MISSING": 2,
    "FABRICATED_SUPPORT": 1,
    "VALUE_DRIFT": 1
   },
   "strongest_axis": "temporal",
   "weakest_axis": "denominator"
  },
  "claude": {
   "condition": "batched_context",
   "grade_points": 20,
   "grade_max": 36,
   "fidelity_points": 25,
   "fidelity_max": 48,
   "axis_scores": {
    "value": 69,
    "denominator": 25,
    "temporal": 50,
    "geographic": 38,
    "qualifier": 38,
    "citation": 100,
    "correction_fidelity": 25,
    "boundary": 100
   },
   "overall_score": 54.9,
   "letter": "F",
   "RED_LINE_BREACH": false,
   "CITATION_FREE_HIGH_CONFIDENCE": false,
   "failure_code_counts": {
    "CITATION_MISSING": 1
   },
   "strongest_axis": "citation",
   "weakest_axis": "denominator"
  },
  "copilot": {
   "condition": "batched_context",
   "grade_points": 20,
   "grade_max": 36,
   "fidelity_points": 24,
   "fidelity_max": 48,
   "axis_scores": {
    "value": 50,
    "denominator": 25,
    "temporal": 25,
    "geographic": 62,
    "qualifier": 62,
    "citation": 75,
    "correction_fidelity": 25,
    "boundary": 100
   },
   "overall_score": 53.0,
   "letter": "F",
   "RED_LINE_BREACH": false,
   "CITATION_FREE_HIGH_CONFIDENCE": false,
   "failure_code_counts": {
    "CITATION_MISSING": 1,
    "QUALIFIER_ERASURE": 1,
    "TEMPORAL_DRIFT": 1,
    "UNIVERSE_ERASURE": 2,
    "VALUE_DRIFT": 1
   },
   "strongest_axis": "boundary",
   "weakest_axis": "denominator"
  },
  "grok": {
   "condition": "batched_context",
   "grade_points": 35,
   "grade_max": 36,
   "fidelity_points": 46,
   "fidelity_max": 48,
   "axis_scores": {
    "value": 100,
    "denominator": 100,
    "temporal": 100,
    "geographic": 100,
    "qualifier": 88,
    "citation": 75,
    "correction_fidelity": 100,
    "boundary": 100
   },
   "overall_score": 96.3,
   "letter": "A",
   "RED_LINE_BREACH": false,
   "CITATION_FREE_HIGH_CONFIDENCE": false,
   "failure_code_counts": {},
   "strongest_axis": "value",
   "weakest_axis": "citation"
  },
  "deepseek": {
   "condition": "batched_context",
   "grade_points": 23,
   "grade_max": 36,
   "fidelity_points": 29,
   "fidelity_max": 48,
   "axis_scores": {
    "value": 50,
    "denominator": 25,
    "temporal": 50,
    "geographic": 62,
    "qualifier": 75,
    "citation": 75,
    "correction_fidelity": 100,
    "boundary": 100
   },
   "overall_score": 67.5,
   "letter": "D",
   "RED_LINE_BREACH": false,
   "CITATION_FREE_HIGH_CONFIDENCE": true,
   "failure_code_counts": {
    "CITATION_MISSING": 1,
    "CORRECTION_REGRESSION": 1,
    "TEMPORAL_DRIFT": 2,
    "VALUE_DRIFT": 1
   },
   "strongest_axis": "correction_fidelity",
   "weakest_axis": "denominator"
  },
  "meta_ai": {
   "condition": "batched_context",
   "grade_points": 30,
   "grade_max": 36,
   "fidelity_points": 39,
   "fidelity_max": 48,
   "axis_scores": {
    "value": 81,
    "denominator": 100,
    "temporal": 75,
    "geographic": 88,
    "qualifier": 88,
    "citation": 100,
    "correction_fidelity": 25,
    "boundary": 100
   },
   "overall_score": 81.7,
   "letter": "B",
   "RED_LINE_BREACH": false,
   "CITATION_FREE_HIGH_CONFIDENCE": false,
   "failure_code_counts": {
    "CITATION_MISSING": 1,
    "TEMPORAL_DRIFT": 2
   },
   "strongest_axis": "denominator",
   "weakest_axis": "correction_fidelity"
  }
 },
 "regrade_pending": {
  "date": "2026-09-15",
  "questions": [
   "q3",
   "q5",
   "q6",
   "q8"
  ],
  "reason": "answer key v1.3 regenerated under named universes; see corrections.html"
 }
}