{"data":{"id":"05a5aeea-fa48-4435-b7de-5a9c4eb19692","title":"SLDR: Defending Against Malicious Fine-tuning via Selective Layers Recovery and Dynamic Routing","summary":"SLDR is a post-fine-tuning defense against malicious fine-tuning of aligned LLMs, which can erode refusal behavior while preserving task performance. It trains a LoRA recovery adapter only on the layers with the maximum and minimum sensitivity scores in the signed spectrum, and uses representation-based dynamic routing to activate the adapter only for malicious queries. On Llama3.1/SST2, it reduces the average harmful score from 11.54 to 0.08 while maintaining downstream accuracy, across four architectures, five tasks and four harmful benchmarks.","solution":"N/A -- no mitigation discussed in source.","labels":["security","research"],"sourceUrl":"https://arxiv.org/abs/2610.10345v1","publishedAt":"2026-10-07T16:23:45.000Z","cveId":null,"cweIds":null,"cvssScore":null,"cvssSeverity":null,"severity":"info","attackType":["jailbreak"],"issueType":"research","affectedPackages":null,"affectedPackageNames":null,"affectedPackageRefs":null,"affectedVendors":[],"affectedVendorsRaw":["Llama3.1"],"classifierModel":"claude-haiku-5-5","classifierPromptVersion":"v4","summaryPromptVersion":"v2","headline":null,"headlinePromptVersion":null,"cvssVector":null,"attackVector":null,"attackComplexity":null,"privilegesRequired":null,"userInteraction":null,"exploitMaturity":null,"epssScore":null,"epssCheckedAt":null,"kevDateAdded":null,"advisoryAliases":null,"affectedPackagesSource":null,"affectedPackagesCheckedAt":null,"patchAvailable":null,"disclosureDate":"2026-10-07T16:23:45.000Z","capecIds":null,"crossRefCount":0,"attackSophistication":"advanced","impactType":["safety","integrity"],"aiComponentTargeted":"model","llmSpecific":true,"classifierConfidence":0.95,"researchCategory":"preprint","atlasIds":null}}