{"data":{"id":"0f529f2d-5868-4cc6-86a5-d078d0befe85","title":"Detect and Suppress: A Mechanistic Defense against Adversarial Patches in VLA Models","summary":"Researchers analyze Vision-Language-Action (VLA) models with a sparse autoencoder (SAE) and find an internal feature whose activation strongly correlates with adversarial patches. They suppress this feature at inference time only when a linear probe detects an attack, which improves robustness without fine-tuning the VLA. On LIBERO-10, conditional intervention raises success rate under intermittent attacks, while continuous intervention substantially degrades policy performance.","solution":"Suppress the identified SAE feature at inference time, applying the intervention only when a linear probe detects an attack. Avoid continuous application, which substantially degrades policy performance.","labels":["security","research"],"sourceUrl":"https://arxiv.org/abs/2610.03498v1","publishedAt":"2026-10-02T15:57:04.000Z","cveId":null,"cweIds":null,"cvssScore":null,"cvssSeverity":null,"severity":"info","attackType":["model_evasion"],"issueType":"research","affectedPackages":null,"affectedPackageNames":null,"affectedPackageRefs":null,"affectedVendors":[],"affectedVendorsRaw":["Vision-Language-Action (VLA) models"],"classifierModel":"claude-haiku-5-5","classifierPromptVersion":"v4","summaryPromptVersion":"v2","headline":null,"headlinePromptVersion":null,"cvssVector":null,"attackVector":null,"attackComplexity":null,"privilegesRequired":null,"userInteraction":null,"exploitMaturity":null,"epssScore":null,"epssCheckedAt":null,"kevDateAdded":null,"advisoryAliases":null,"affectedPackagesSource":null,"affectedPackagesCheckedAt":null,"patchAvailable":null,"disclosureDate":"2026-10-02T15:57:04.000Z","capecIds":null,"crossRefCount":0,"attackSophistication":"advanced","impactType":["integrity","safety"],"aiComponentTargeted":"model","llmSpecific":false,"classifierConfidence":0.95,"researchCategory":"preprint","atlasIds":null}}