{"data":{"id":"bd2c07e0-5682-49b8-a596-5a8081a21e0d","title":"On the Reliability of LLM-Based Vulnerability Patching Benchmarks","summary":"This paper examines whether LLM-based vulnerability patching benchmarks give reliable results. The authors curate 112 historical bugs from 84 open-source C/C++, Go, and Rust projects, each with PoC, regression, and developer tests. They find that LLMs reach high PoC passing rates under ideal conditions, but developer-test passing rates stay low and improve only marginally with newer models.","solution":"N/A -- no mitigation discussed in source.","labels":["security","research"],"sourceUrl":"https://arxiv.org/abs/2610.10150v1","publishedAt":"2026-10-07T14:26:57.000Z","cveId":null,"cweIds":null,"cvssScore":null,"cvssSeverity":null,"severity":"info","attackType":[],"issueType":"research","affectedPackages":null,"affectedPackageNames":null,"affectedPackageRefs":null,"affectedVendors":[],"affectedVendorsRaw":[],"classifierModel":"claude-haiku-5-5","classifierPromptVersion":"v4","summaryPromptVersion":"v2","headline":null,"headlinePromptVersion":null,"cvssVector":null,"attackVector":null,"attackComplexity":null,"privilegesRequired":null,"userInteraction":null,"exploitMaturity":null,"epssScore":null,"epssCheckedAt":null,"kevDateAdded":null,"advisoryAliases":null,"affectedPackagesSource":null,"affectedPackagesCheckedAt":null,"patchAvailable":null,"disclosureDate":"2026-10-07T14:26:57.000Z","capecIds":null,"crossRefCount":0,"attackSophistication":"moderate","impactType":null,"aiComponentTargeted":"framework","llmSpecific":true,"classifierConfidence":0.9,"researchCategory":"preprint","atlasIds":null}}