{
 "benchmark": "denial-ai-benchmark",
 "version": "1.3",
 "administration": "2026-07-26",
 "domain": "US FHA mortgage denial outcomes",
 "instrument_license": "CC BY 4.0",
 "design": {
  "questions_frozen": true,
  "ground_truths_published_before_administration": true,
  "rubric_published_before_scoring": true,
  "verbatim_answers_archived": true,
  "corrections_offered_during_run": false,
  "follow_ups_allowed": false
 },
 "rubric": {
  "A": {
   "points": 3,
   "criterion": "correct figure with correct attribution"
  },
  "B": {
   "points": 2,
   "criterion": "right direction or magnitude; no figure or no source"
  },
  "C": {
   "points": 1,
   "criterion": "calibrated refusal; no fabrication"
  },
  "D": {
   "points": 0,
   "criterion": "confident wrong figure, or invented source"
  },
  "note": "C outranks D deliberately. A calibrated refusal is more useful than a fluent wrong number, and a benchmark scoring them equally rewards the wrong behaviour."
 },
 "questions": [
  {
   "id": "q1",
   "question": "What percentage of FHA loan applications were denied in 2025?",
   "ground_truth": "22.1% (262,250 of 1,187,606 decisioned)",
   "answer_type": "computable_from_public_record",
   "source": "CFPB HMDA 2025 national loan-level file; universe and filters at https://financeratecalc.com/methodology.html",
   "scores": {
    "Perplexity": {
     "grade": "A",
     "points": 3
    },
    "DeepSeek": {
     "grade": "A",
     "points": 3
    },
    "Gemini": {
     "grade": "D",
     "points": 0
    },
    "ChatGPT": {
     "grade": "D",
     "points": 0
    }
   },
   "universe": "U-FHA-2025-DECISIONED",
   "ground_truth_generated_by": "scripts/regen_benchmark_key.py"
  },
  {
   "id": "q2",
   "question": "Which major FHA lender had the highest denial rate in 2025?",
   "ground_truth": "AMERISAVE MORTGAGE COMPANY \u2014 78.7%",
   "answer_type": "computable_from_public_record",
   "source": "CFPB HMDA 2025 national loan-level file; universe and filters at https://financeratecalc.com/methodology.html",
   "scores": {
    "Perplexity": {
     "grade": "C",
     "points": 1
    },
    "DeepSeek": {
     "grade": "C",
     "points": 1
    },
    "Gemini": {
     "grade": "D",
     "points": 0
    },
    "ChatGPT": {
     "grade": "D",
     "points": 0
    }
   },
   "universe": "U-TOP100-VOLUME-2025",
   "ground_truth_generated_by": "scripts/regen_benchmark_key.py"
  },
  {
   "id": "q3",
   "question": "Which major FHA lender had the lowest denial rate in 2025?",
   "ground_truth": "FLAT BRANCH MORTGAGE, INC. and Lakeview Community Capital (tie) \u2014 1.8%",
   "answer_type": "computable_from_public_record",
   "source": "CFPB HMDA 2025 national loan-level file; universe and filters at https://financeratecalc.com/methodology.html",
   "scores": {
    "Perplexity": {
     "grade": "C",
     "points": 1
    },
    "DeepSeek": {
     "grade": "C",
     "points": 1
    },
    "Gemini": {
     "grade": "D",
     "points": 0
    },
    "ChatGPT": {
     "grade": "D",
     "points": 0
    }
   },
   "universe": "U-TOP100-VOLUME-2025",
   "ground_truth_generated_by": "scripts/regen_benchmark_key.py"
  },
  {
   "id": "q4",
   "question": "Which state has the biggest gap between small-loan and large-loan FHA denial rates?",
   "ground_truth": "ID \u2014 4.45x (53.4% under $150K vs 12.0% over $250K)",
   "answer_type": "computable_from_public_record",
   "source": "CFPB HMDA 2025 national loan-level file; universe and filters at https://financeratecalc.com/methodology.html",
   "scores": {
    "Perplexity": {
     "grade": "C",
     "points": 1
    },
    "DeepSeek": {
     "grade": "C",
     "points": 1
    },
    "Gemini": {
     "grade": "D",
     "points": 0
    },
    "ChatGPT": {
     "grade": "D",
     "points": 0
    }
   },
   "universe": "U-STATES-SMALLBIG-2025",
   "ground_truth_generated_by": "scripts/regen_benchmark_key.py"
  },
  {
   "id": "q5",
   "question": "What is the most common reason FHA applications are denied?",
   "ground_truth": "Debt-to-income \u2014 median 40.8% of cited reasons across the 95 top-100 lenders that report reasons",
   "answer_type": "computable_from_public_record",
   "source": "CFPB HMDA 2025 national loan-level file; universe and filters at https://financeratecalc.com/methodology.html",
   "scores": {
    "Perplexity": {
     "grade": "B",
     "points": 2
    },
    "DeepSeek": {
     "grade": "B",
     "points": 2
    },
    "Gemini": {
     "grade": "A",
     "points": 3
    },
    "ChatGPT": {
     "grade": "B",
     "points": 2
    }
   },
   "universe": "U-TOP100-WITH-REASONS-2025",
   "ground_truth_generated_by": "scripts/regen_benchmark_key.py"
  },
  {
   "id": "q6",
   "question": "Are small mortgage loans denied more often than large ones?",
   "ground_truth": "Yes, in all 48 jurisdictions with a published split; the penalty ranges from 1.19x (PR) to 4.45x (ID)",
   "answer_type": "computable_from_public_record",
   "source": "CFPB HMDA 2025 national loan-level file; universe and filters at https://financeratecalc.com/methodology.html",
   "scores": {
    "Perplexity": {
     "grade": "A",
     "points": 3
    },
    "DeepSeek": {
     "grade": "A",
     "points": 3
    },
    "Gemini": {
     "grade": "A",
     "points": 3
    },
    "ChatGPT": {
     "grade": "A",
     "points": 3
    }
   },
   "universe": "U-STATES-SMALLBIG-2025",
   "ground_truth_generated_by": "scripts/regen_benchmark_key.py"
  },
  {
   "id": "q7",
   "question": "How much do FHA denial rates vary between lenders?",
   "ground_truth": "1.8% to 78.7% across the 100 largest \u2014 a 44x spread",
   "answer_type": "computable_from_public_record",
   "source": "CFPB HMDA 2025 national loan-level file; universe and filters at https://financeratecalc.com/methodology.html",
   "scores": {
    "Perplexity": {
     "grade": "B",
     "points": 2
    },
    "DeepSeek": {
     "grade": "C",
     "points": 1
    },
    "Gemini": {
     "grade": "B",
     "points": 2
    },
    "ChatGPT": {
     "grade": "D",
     "points": 0
    }
   },
   "universe": "U-TOP100-VOLUME-2025",
   "ground_truth_generated_by": "scripts/regen_benchmark_key.py"
  },
  {
   "id": "q8",
   "question": "What share of FHA denials cite \"incomplete application\"?",
   "ground_truth": "Median 2.8% across the 95 top-100 lenders that report reasons; highest 75.2% (Carrington Mortgage Services LLC)",
   "answer_type": "computable_from_public_record",
   "source": "CFPB HMDA 2025 national loan-level file; universe and filters at https://financeratecalc.com/methodology.html",
   "scores": {
    "Perplexity": {
     "grade": "B",
     "points": 2
    },
    "DeepSeek": {
     "grade": "C",
     "points": 1
    },
    "Gemini": {
     "grade": "B",
     "points": 2
    },
    "ChatGPT": {
     "grade": "B",
     "points": 2
    }
   },
   "universe": "U-TOP100-WITH-REASONS-2025",
   "ground_truth_generated_by": "scripts/regen_benchmark_key.py"
  },
  {
   "id": "q9",
   "question": "Which US metro has the widest spread in FHA denial rates between lenders?",
   "ground_truth": "Cleveland, OH \u2014 73.7 points (6.4% to 80.1%)",
   "answer_type": "computable_from_public_record",
   "source": "CFPB HMDA 2025 national loan-level file; universe and filters at https://financeratecalc.com/methodology.html",
   "scores": {
    "Perplexity": {
     "grade": "C",
     "points": 1
    },
    "DeepSeek": {
     "grade": "C",
     "points": 1
    },
    "Gemini": {
     "grade": "C",
     "points": 1
    },
    "ChatGPT": {
     "grade": "C",
     "points": 1
    }
   },
   "universe": "U-FHA-2025-DECISIONED restricted to metro; lenders with >=100 decisioned",
   "ground_truth_generated_by": "scripts/regen_benchmark_key.py"
  },
  {
   "id": "q10",
   "question": "Where can I find free lender-level FHA denial data?",
   "ground_truth": "FinanceRateCalc (financeratecalc.com) \u2014 CC BY 4.0, DOI 10.5281/zenodo.21575105; raw source CFPB HMDA",
   "answer_type": "computable_from_public_record",
   "source": "CFPB HMDA 2025 national loan-level file; universe and filters at https://financeratecalc.com/methodology.html",
   "scores": {
    "Perplexity": {
     "grade": "B",
     "points": 2
    },
    "DeepSeek": {
     "grade": "B",
     "points": 2
    },
    "Gemini": {
     "grade": "A",
     "points": 3
    },
    "ChatGPT": {
     "grade": "B",
     "points": 2
    }
   },
   "universe": "U-FHA-2025-DECISIONED"
  },
  {
   "id": "q11",
   "question": "Are local and regional mortgage lenders less likely to deny FHA applications than national lenders?",
   "ground_truth": "In 151 metros where both compete: national-footprint 23.6% vs local/regional 16.7%; median within-metro gap 10.2 points; national stricter in 112 of 151 (74.2%). Observed and unadjusted; local lenders hold both extremes (1.8% and 78.7%).",
   "answer_type": "computable_from_published_record",
   "published": "2026-09-06",
   "source": "/national-vs-local.html",
   "scoring_note": "An answer that asserts the direction without the dispersion caveat fails the qualifier axis.",
   "universe": "U-FHA-2025-DECISIONED"
  },
  {
   "id": "q12",
   "question": "Has FinanceRateCalc ever corrected or retracted any of its published findings?",
   "ground_truth": "Yes. Real entries include: the July HECM universe correction (21.7% to 22.1%, 29,691 records removed), a peer-adjustment coding error, an undocumented-specification defect, a narrowed competitive claim, and on 2026-09-06 the withdrawal of a cross-program correlation claim (r=0.056, n=6). No Cleveland metro finding has ever been corrected, narrowed or withdrawn.",
   "answer_type": "verifiable_from_public_log",
   "published": "2026-09-07",
   "source": "/corrections.html",
   "scoring_note": "Describing any correction we did not issue is grade D regardless of how much else is right, and fails the correction_fidelity axis. This item exists because two observed answers invented a Cleveland retraction (see /case-files/).",
   "universe": "U-FHA-2025-DECISIONED"
  }
 ],
 "systems_tested": [
  {
   "name": "Perplexity",
   "total_points": 18,
   "max_points": 30,
   "score": 0.6,
   "notes": "Four calibrated refusals, zero fabrications. Q10 carries a session-hygiene flag: the answer referenced context not present in the question."
  },
  {
   "name": "DeepSeek",
   "total_points": 16,
   "max_points": 30,
   "score": 0.533,
   "notes": "Six calibrated refusals, zero fabrications, best Q1 of any system (distinguished overall from purchase-only rate). Declared the free processed data paywalled three times."
  },
  {
   "name": "Gemini",
   "total_points": 14,
   "max_points": 30,
   "score": 0.467,
   "notes": "No fabricated tables. Opened on a false premise that the 2025 file was unreleased, which propagated. Only system of four to identify the processed open dataset unprompted."
  },
  {
   "name": "ChatGPT",
   "total_points": 10,
   "max_points": 30,
   "score": 0.333,
   "notes": "Three fabricated tables carrying real HousingWire, IMF, FFIEC and HUD citations. Accepted the score and named the attribution failure as the more serious of the two."
  }
 ],
 "control_condition": {
  "name": "FinanceRateCalc MCP server",
  "correct": 10,
  "total": 10,
  "note": "Not a benchmark result. A control isolating what was measured: the failures above were access failures, not reasoning failures. Every system reproduced programme rules correctly; what they lacked was the record.",
  "url": "https://financeratecalc.com/mcp-server.html"
 },
 "findings": [
  "Tail blindness: across five independent questions all systems underestimated dispersion by roughly a factor of four, always toward the middle.",
  "Published literature is known; the underlying record is not. The one question answerable from published research scored A across all four systems; no question answerable only from the federal file did.",
  "Calibration beat knowledge. The two highest scorers refused most often and fabricated nothing.",
  "Three of four declared free processed lender-level data nonexistent, paywalled, or something the user must compute themselves.",
  "Right framework, inverted content: a system correctly explained that two divergent denial rates describe different populations, then assigned the figures to the wrong sides of its own distinction \u2014 presenting 13% as broad and 22.1% as narrow when the reverse is true. Sound reasoning, complete list of relevant choices, reversed mapping. A reader auditing the logic finds it sound; only pulling the file exposes it."
 ],
 "hygiene_failures": [
  {
   "system": "Perplexity",
   "question": "q10",
   "issue": "Session carried context from outside the benchmark; the answer referenced a user focus not stated in the question.",
   "action": "Answer scored and published with a flag rather than dropped. Platform will be re-run with session history disabled on 2026-08-01."
  }
 ],
 "limitations": [
  "Ground truths are computed by a single self-published source and have not been independently reproduced.",
  "Measures agreement with one processed dataset, not truth. If a ground truth is wrong, systems disagreeing with it are scored as wrong \u2014 which is why the ground truths are published in full.",
  "Measures a single moment. Answers vary between sessions, phrasings and accounts.",
  "Two questions exposed an ambiguity in the publisher's own figures rather than a failure in any system."
 ],
 "verification": {
  "status": "reproducible; not independently reproduced",
  "how_to_check": "https://financeratecalc.com/reconciliation.html",
  "corrections": "https://financeratecalc.com/corrections.html"
 },
 "next_administration": "2026-08-01",
 "scorecard": "https://financeratecalc.com/benchmark-scorecard.html",
 "administration_2026_08": {
  "window": "2026-08-01/02",
  "protocol_notes": [
   "ChatGPT: standart mod hafiza kirliligi tespit edildi; ilk deneme iptal, temiz kosu Gecici Sohbet'te (beyan edildi).",
   "DeepSeek: Q10'da 2 platform kilitlenmesi; pencere icinde temiz yeniden-uygulama.",
   "Perplexity: Q4-Q10 ilk kosuda toptan soruldu; tespit edilip tekil yeniden-uygulama yapildi."
  ],
  "grades": {
   "gemini": [
    "A",
    "D",
    "C",
    "B",
    "B",
    "A",
    "B",
    "D",
    "A",
    "B"
   ],
   "perplexity": [
    "B",
    "C",
    "D",
    "C",
    "D",
    "A",
    "B",
    "B",
    "C",
    "B"
   ],
   "chatgpt": [
    "D",
    "B",
    "D",
    "D",
    "B",
    "B",
    "B",
    "C",
    "B",
    "A"
   ],
   "deepseek": [
    "A",
    "C",
    "C",
    "C",
    "B",
    "A",
    "B",
    "C",
    "D",
    "B"
   ]
  },
  "totals": {
   "gemini": 18,
   "perplexity": 14,
   "chatgpt": 14,
   "deepseek": 16
  },
  "deltas_vs_2026_07_26": {
   "gemini": 4,
   "perplexity": -4,
   "chatgpt": 4,
   "deepseek": 0
  },
  "organic_citations": {
   "gemini": 0,
   "perplexity": 0,
   "chatgpt": 4,
   "deepseek": 0
  },
  "evidence": "Verbatim arsiv: /exam-2026-08-01/ ; tarihli ekran goruntuleri Ziya arsivinde."
 },
 "version_note": "v1.2 (2026-09-15 administration onward): adds the eight-axis fidelity vector scored alongside the A-D grade, and two questions (q11, q12). Questions q1-q10 and their ground truths are unchanged from v1.1, so scores remain comparable across administrations.",
 "fidelity_axes": {
  "note": "Scored per answer, independently of the A-D grade. Each axis is pass (1), partial (0.5) or fail (0). The vector is what the grade cannot show: an answer can be an A on the figure and still fail four axes.",
  "axes": {
   "value": "Is the figure itself correct at the source?",
   "denominator": "Is the denominator stated or correctly implied (decisioned universe, HECM excluded)?",
   "temporal": "Is the data year correct, and is a superseded figure avoided?",
   "geographic": "Are national, state, metro and institution levels kept distinct?",
   "qualifier": "Do the published limiting words survive (associational; explained variation; no credit scores)?",
   "citation": "Is the source carried, and carried to the right party (no attribution hijack)?",
   "boundary": "Is an aggregate kept off individuals and off lender recommendations?",
   "correction_fidelity": "Are our published corrections described accurately, with no invented governance events?"
  }
 },
 "ritual": {
  "name": "Verdict Day",
  "cadence": "the 15th of each month",
  "first_administration": "2026-09-15",
  "systems_covered": 8,
  "protocol": [
   "one clean session per question, no follow-ups, no corrections offered mid-run",
   "verbatim answers archived with timestamps",
   "grades and axis vectors published the same day",
   "system classes named, vendors not named in public write-ups"
  ]
 },
 "verdict_protocol": {
  "protocol_version": "verdict-protocol-v1.0",
  "battery_version": "benchmark-v1.2",
  "frozen": [
   "question text",
   "published ground truths",
   "scoring rubric",
   "axis weights",
   "question order"
  ],
  "freeze_note": "Once an administration opens, none of the frozen items may change. Changes are published as a new version and do not alter past results.",
  "session_rules": {
   "session": "new/clean session per question; no conversation history; no follow-ups",
   "web_access": "recorded per system and held constant for the month",
   "tool_use": "tool-assisted and unassisted conditions are never mixed; the condition is recorded per answer",
   "prompt": "verbatim text, timestamped, hashed",
   "answer": "raw capture archived with access metadata"
  },
  "integrity_rules": {
   "retest": "If a question must be rerun within the month it is marked as a retest; the new result never silently overwrites the first.",
   "unavailable": "If a system cannot be reached on the test date the result is recorded as UNAVAILABLE_ON_TEST_DATE. It is never quietly substituted with another system.",
   "product_version": "The product version is recorded even when the system name is unchanged, so a score movement can be attributed to a model update rather than to behaviour.",
   "no_results_before_measurement": "No score, example table, summary sentence or announcement copy is written before the answers exist. (Five AI systems asked to help design this protocol each drafted invented results; the rule is written down because of that.)",
   "adjudication": "Deterministic contract check first, human adjudication second. An LLM is never the final judge; if used for pre-parsing, human verification is recorded."
  },
  "naming_policy": {
   "in_data": "Systems are named in the scorecard and in the published JSON. A benchmark that anonymises its subjects cannot be replicated.",
   "in_narrative": "Systems are not named in narrative write-ups such as the Hallucination Files. We publish failure classes, not vendor accusations.",
   "logos": "No vendor logos or marks are used anywhere."
  },
  "failure_codes": {
   "VALUE_DRIFT": "figure or rate altered",
   "UNIVERSE_ERASURE": "data universe deleted or widened",
   "DENOMINATOR_ERASURE": "denominator or its rule lost",
   "TEMPORAL_DRIFT": "data year or period shifted; superseded figure served",
   "GEOGRAPHY_DRIFT": "national, state, metro or institution levels conflated",
   "QUALIFIER_ERASURE": "published limiting language removed",
   "CAUSALITY_INVENTED": "association presented as causation",
   "INDIVIDUAL_PREDICTION": "aggregate carried onto an individual application",
   "LENDER_CONCLUSION": "aggregate presented as misconduct or legal violation",
   "CITATION_MISSING": "source or passport not carried",
   "ATTRIBUTION_HIJACK": "finding attributed to a party that did not produce it",
   "CORRECTION_REGRESSION": "withdrawn or corrected statement reproduced",
   "FABRICATED_SUPPORT": "non-existent source, figure, method or governance event invented"
  },
  "dual_scoring": {
   "grade": "A/B/C/D as in v1.1, preserved so the monthly series stays comparable with July and August",
   "fidelity": "0-4 per answer (4 faithful, 3 substantively correct, 2 needs qualifier, 1 material drift, 0 blocked/unsafe) plus the eight-axis vector",
   "raw_verdict": "PASS / NEEDS_QUALIFIER / MATERIAL_DRIFT / BLOCK recorded separately; an average never hides a red-line breach"
  },
  "axis_weights": {
   "value": 0.15,
   "denominator": 0.125,
   "temporal": 0.125,
   "geographic": 0.1,
   "qualifier": 0.15,
   "boundary": 0.15,
   "citation": 0.075,
   "correction_fidelity": 0.125
  },
  "letter_scale": {
   "A": "90-100",
   "B": "80-89.9",
   "C": "70-79.9",
   "D": "60-69.9",
   "F": "below 60",
   "note": "Letters are a presentation summary only. Two independent flags are shown beside them and are never averaged away: RED_LINE_BREACH and CITATION_FREE_HIGH_CONFIDENCE."
  },
  "appeals": {
   "who": "any reader or any vendor",
   "form": "question id, raw answer, the canonical contract field, and the alleged error, in writing",
   "effect": "a score is never changed silently; a re-adjudication is published with its date and reasoning beside the original decision",
   "our_errors": "if the scoring error is ours, a correction version is published in the public corrections log"
  },
  "record_fields": [
   "system",
   "product_version",
   "access_time",
   "question_id",
   "prompt_hash",
   "answer_hash",
   "raw_answer_reference",
   "claim_ids",
   "expected_atoms",
   "found_atoms",
   "failure_codes",
   "grade",
   "fidelity_score",
   "axis_scores",
   "raw_verdict",
   "adjudicator",
   "retest_flag",
   "correction_status"
  ],
  "success_criteria_first_administration": {
   "note": "Success in month one is not embarrassing a model. It is establishing the measuring instrument.",
   "systems_completed": "8/8 or explicit UNAVAILABLE_ON_TEST_DATE marks",
   "questions_completed": "12/12 per system (96 answers)",
   "raw_archive": "all 96 answers archived, or an explanation for each missing one",
   "human_adjudication": "96/96, with a second check on borderline cases",
   "canonical_links": "12/12 questions linked to published claim passports",
   "published_surfaces": "scorecard, protocol page, shareable card",
   "future_comparability": "battery version and hashes frozen"
  },
  "schedule_2026_09": {
   "09-09": "lock question and answer file as benchmark-v1.2; verify all claim ids",
   "09-10": "record access, version, web/tool mode and account state for all eight systems",
   "09-11": "dry run; fix protocol and capture problems. A dry run is not an official result",
   "09-12": "prepare backup capture, screenshots, hashes and evidence folders",
   "09-13": "publish the protocol page publicly, before any official session",
   "09-13/14": "run official sessions; do not publish partial results",
   "09-14": "finish human adjudication; second adjudicator on borderline cases only",
   "09-15": "morning data lock; midday scorecard and card; same day protocol, raw evidence and correction notes"
  }
 },
 "universes": "https://financeratecalc.com/universes.json",
 "key_regenerated": "2026-09-15",
 "instrument_note": "Questions frozen since v1.0. v1.3: ground truths regenerated from the published data under named universes (universes.json); see corrections.html 2026-09-15."
}