{"data":{"id":"8e7c570f-83f7-48cd-aed1-08015f29bff9","title":"PatchBench: Measuring Collateral Damage in Activation Patching","summary":"PatchBench introduces a benchmark of 400 curated, model-specific jailbreak failures drawn from 27,870 prompts across 37 public datasets, filtered with WildGuard, pairwise Elo ranking and manual verification. Its companion protocol, PatchBench-Local, tests whether a patch is behaviourally precise by checking harmful-neighbour correction and benign-neighbour preservation. Evaluating four activation steering methods, the authors found global capability could stay nearly unchanged while local benign regressions were severe, showing aggregate metrics miss collateral damage.","solution":"N/A -- no mitigation discussed in source.","labels":["research","safety"],"sourceUrl":"https://arxiv.org/abs/2610.10276v1","publishedAt":"2026-10-07T15:43:28.000Z","cveId":null,"cweIds":null,"cvssScore":null,"cvssSeverity":null,"severity":"info","attackType":["jailbreak"],"issueType":"research","affectedPackages":null,"affectedPackageNames":null,"affectedPackageRefs":null,"affectedVendors":[],"affectedVendorsRaw":["open-source instruction-tuned LLMs"],"classifierModel":"claude-haiku-5-5","classifierPromptVersion":"v4","summaryPromptVersion":"v2","headline":null,"headlinePromptVersion":null,"cvssVector":null,"attackVector":null,"attackComplexity":null,"privilegesRequired":null,"userInteraction":null,"exploitMaturity":null,"epssScore":null,"epssCheckedAt":null,"kevDateAdded":null,"advisoryAliases":null,"affectedPackagesSource":null,"affectedPackagesCheckedAt":null,"patchAvailable":null,"disclosureDate":"2026-10-07T15:43:28.000Z","capecIds":null,"crossRefCount":0,"attackSophistication":"moderate","impactType":["safety"],"aiComponentTargeted":"model","llmSpecific":true,"classifierConfidence":0.95,"researchCategory":"preprint","atlasIds":null}}