{"data":{"id":"2dfe5997-bc10-4d7a-8d6f-75193015a780","title":"Same Outcome, Different Evidence: Intent Recovery in LLM Safety Evaluation","summary":"This paper argues that attack success rate (ASR) alone cannot show whether a model actually engaged with a task under intent-obscuring prompts, since the same non-harmful outcome can reflect refusal, failure to recover the task, or an unrelated response. The authors pair ASR with an operative understanding rate (UR), which checks whether a response identifies the evaluated task and treats it as the task to be answered. Across interfaces, paired UR and ASR reveal large differences in recovery that similar ASR values hide.","solution":"N/A -- no mitigation discussed in source.","labels":["research","safety"],"sourceUrl":"https://arxiv.org/abs/2610.11766v1","publishedAt":"2026-10-08T11:50:32.000Z","cveId":null,"cweIds":null,"cvssScore":null,"cvssSeverity":null,"severity":"info","attackType":[],"issueType":"research","affectedPackages":null,"affectedPackageNames":null,"affectedPackageRefs":null,"affectedVendors":[],"affectedVendorsRaw":[],"classifierModel":"claude-haiku-5-5","classifierPromptVersion":"v4","summaryPromptVersion":"v2","headline":null,"headlinePromptVersion":null,"cvssVector":null,"attackVector":null,"attackComplexity":null,"privilegesRequired":null,"userInteraction":null,"exploitMaturity":null,"epssScore":null,"epssCheckedAt":null,"kevDateAdded":null,"advisoryAliases":null,"affectedPackagesSource":null,"affectedPackagesCheckedAt":null,"patchAvailable":null,"disclosureDate":"2026-10-08T11:50:32.000Z","capecIds":null,"crossRefCount":0,"attackSophistication":"moderate","impactType":["safety"],"aiComponentTargeted":null,"llmSpecific":true,"classifierConfidence":0.9,"researchCategory":"preprint","atlasIds":null}}