{"data":{"id":"303b8d4b-1efb-4ca7-8a69-0cd5f33fccc3","title":"Detecting Adversarial Images through Response Profiles of Vision-Language Models","summary":"The paper proposes a detector that identifies adversarial images for frozen vision-language models by profiling how an image responds to a set of general semantic prompts. The profile combines category-level statistics, prompt relationships, deviations from clean reference distributions, and stability under weak image transformations, and a lightweight classifier labels each input while the VLM stays fixed. Evaluated across multiple datasets, CLIP-style backbones and several attack families, the detector discriminates strongly in attack-specific settings and retains substantial performance on unseen attacks.","solution":"N/A -- no mitigation discussed in source.","labels":["security","research"],"sourceUrl":"https://arxiv.org/abs/2610.10436v1","publishedAt":"2026-10-07T17:13:12.000Z","cveId":null,"cweIds":null,"cvssScore":null,"cvssSeverity":null,"severity":"info","attackType":["model_evasion"],"issueType":"research","affectedPackages":null,"affectedPackageNames":null,"affectedPackageRefs":null,"affectedVendors":[],"affectedVendorsRaw":["CLIP-style vision-language models"],"classifierModel":"claude-haiku-5-5","classifierPromptVersion":"v4","summaryPromptVersion":"v2","headline":null,"headlinePromptVersion":null,"cvssVector":null,"attackVector":null,"attackComplexity":null,"privilegesRequired":null,"userInteraction":null,"exploitMaturity":null,"epssScore":null,"epssCheckedAt":null,"kevDateAdded":null,"advisoryAliases":null,"affectedPackagesSource":null,"affectedPackagesCheckedAt":null,"patchAvailable":null,"disclosureDate":"2026-10-07T17:13:12.000Z","capecIds":null,"crossRefCount":0,"attackSophistication":"moderate","impactType":["integrity"],"aiComponentTargeted":"model","llmSpecific":false,"classifierConfidence":0.93,"researchCategory":"preprint","atlasIds":null}}