{"data":{"id":"dd789097-9e99-4539-b37a-c0cb2cabd4bc","title":"Transferable Spatial Temporal Coherence Adversarial Attack on Black-Box Vision Language Models for Autonomous Driving","summary":"Researchers introduce STCA (Spatial Temporal Coherence Adversarial Attack), a black-box method that perturbs driving video to fool Vision Language Models. The attack selects semantically important frames with caption guidance, applies a spatial perturbation that preserves high SSIM, then disrupts cross-frame temporal coherence with a motion-guided mask. Tested on BDD100K and nuScenes against Video LLaVA-7B, Qwen2.5-VL-7B and Dolphin, the spatial attack reaches a high ASR while keeping SSIM high, showing these models remain highly susceptible.","solution":"N/A -- no mitigation discussed in source.","labels":["security","research"],"sourceUrl":"https://arxiv.org/abs/2610.08331v1","publishedAt":"2026-10-06T13:30:06.000Z","cveId":null,"cweIds":null,"cvssScore":null,"cvssSeverity":null,"severity":"info","attackType":["model_evasion"],"issueType":"research","affectedPackages":null,"affectedPackageNames":null,"affectedPackageRefs":null,"affectedVendors":[],"affectedVendorsRaw":["Video LLaVA-7B","Qwen2.5-VL-7B","Dolphin"],"classifierModel":"claude-haiku-5-5","classifierPromptVersion":"v4","summaryPromptVersion":"v2","headline":null,"headlinePromptVersion":null,"cvssVector":null,"attackVector":null,"attackComplexity":null,"privilegesRequired":null,"userInteraction":null,"exploitMaturity":null,"epssScore":null,"epssCheckedAt":null,"kevDateAdded":null,"advisoryAliases":null,"affectedPackagesSource":null,"affectedPackagesCheckedAt":null,"patchAvailable":null,"disclosureDate":"2026-10-06T13:30:06.000Z","capecIds":null,"crossRefCount":0,"attackSophistication":"advanced","impactType":["integrity","safety"],"aiComponentTargeted":"model","llmSpecific":true,"classifierConfidence":0.93,"researchCategory":"preprint","atlasIds":null}}