{"data":{"id":"f3ed5585-248f-4e78-b373-fe4d4d93fa4f","title":"Negative Results for Sparse Autoencoders On Downstream Tasks and Deprioritising SAE Research…","summary":"Google DeepMind's mechanistic interpretability team tested whether sparse autoencoders (SAEs) help with out-of-distribution detection of harmful intent in user prompts. SAEs underperformed linear probes, which the team found cheap and strong. As a result, the team is deprioritising fundamental SAE research while keeping SAEs as one tool.","solution":"N/A -- no mitigation discussed in source.","labels":["research","safety"],"sourceUrl":"https://deepmindsafetyresearch.medium.com/negative-results-for-sparse-autoencoders-on-downstream-tasks-and-deprioritising-sae-research-6cadcfc125b9?source=rss-55e08ddea42e------2","publishedAt":"2025-03-26T19:06:59.000Z","cveId":null,"cweIds":null,"cvssScore":null,"cvssSeverity":null,"severity":"info","attackType":[],"issueType":"research","affectedPackages":null,"affectedPackageNames":null,"affectedPackageRefs":null,"affectedVendors":["Google"],"affectedVendorsRaw":["Google DeepMind"],"classifierModel":"claude-haiku-5-5","classifierPromptVersion":"v4","summaryPromptVersion":"v2","headline":null,"headlinePromptVersion":null,"cvssVector":null,"attackVector":null,"attackComplexity":null,"privilegesRequired":null,"userInteraction":null,"exploitMaturity":null,"epssScore":null,"epssCheckedAt":null,"kevDateAdded":null,"advisoryAliases":null,"affectedPackagesSource":null,"affectedPackagesCheckedAt":null,"patchAvailable":null,"disclosureDate":"2025-03-26T19:06:59.000Z","capecIds":null,"crossRefCount":0,"attackSophistication":"moderate","impactType":null,"aiComponentTargeted":null,"llmSpecific":true,"classifierConfidence":0.85,"researchCategory":"industry","atlasIds":null}}