{
  "metric": "Compiler acceptance within up to three attempts, with error feedback",
  "sourceRepository": "https://github.com/Morten-Ness/Leaner",
  "sourceCommit": "72e0c5afcc06c757b92ab9c3e26349fbc20574fa",
  "sourceFile": "LeanWorkspace/LeanWorkspace/LLMresponses/results.json",
  "leanVersion": "4.29.0",
  "mathlibVersion": "4.29.0",
  "totalProblems": 1513,
  "models": [
    { "name": "GPT-5.4", "modelId": "gpt-5.4", "status": "recorded", "percentage": 53.6, "passed": 811, "failed": 702, "passesByAttempt": { "1": 435, "2": 242, "3": 134 } }
  ],
  "caveats": [
    "The README failure count is inconsistent. The cached log contains 811 PASS and 702 FAIL entries.",
    "PASS means the solver observed no compiler errors; it does not exclude sorry warnings or enforce statement preservation.",
    "Targets import full Mathlib. The CSV flags 128 passing responses for a textual target-name match. These are included in the recorded score."
  ],
  "datasetExamples": 406,
  "example": { "filename": "Associated.exists_mem_finset_dvd.lean", "status": "PASS", "attempts": 2, "selfReferenceFlag": false }
}
