{"data":{"id":"0148b664-377a-4866-aae1-460ea6526ec3","title":"Exploring backdoor attack and defense algorithms in LLMS: Enhancing in-context learning security","summary":"This paper shows that an attacker can manipulate LLM behavior by poisoning the demonstration context used in in-context learning, without fine-tuning the model. The authors present ICLAttack, a backdoor method that poisons demonstration examples or demonstration prompts, reporting a 95.0% average attack success rate on OPT models across three datasets. They also propose ICLDefense, which uses a lightweight auxiliary model and an ensemble-based strategy to refine LLM outputs and reduce attack success.","solution":"ICLDefense: a defense algorithm that utilizes model ensembles, employing a lightweight auxiliary model to refine LLM outputs through an ensemble-based strategy, which the source says substantially reduces the attack success rate compared to existing methods while preserving model performance.","labels":["security","research"],"sourceUrl":"https://doi.org/10.1016/j.patcog.2026.115005","publishedAt":"2026-09-29T00:00:00.000Z","cveId":null,"cweIds":null,"cvssScore":null,"cvssSeverity":null,"severity":"info","attackType":["jailbreak","other"],"issueType":"research","affectedPackages":null,"affectedPackageNames":null,"affectedPackageRefs":null,"affectedVendors":[],"affectedVendorsRaw":["OPT"],"classifierModel":"claude-haiku-5-5","classifierPromptVersion":"v4","summaryPromptVersion":"v2","headline":null,"headlinePromptVersion":null,"cvssVector":null,"attackVector":null,"attackComplexity":null,"privilegesRequired":null,"userInteraction":null,"exploitMaturity":null,"epssScore":null,"epssCheckedAt":null,"kevDateAdded":null,"advisoryAliases":null,"affectedPackagesSource":null,"affectedPackagesCheckedAt":null,"patchAvailable":null,"disclosureDate":"2026-09-29T00:00:00.000Z","capecIds":null,"crossRefCount":0,"attackSophistication":"moderate","impactType":["integrity","safety"],"aiComponentTargeted":"training_data","llmSpecific":true,"classifierConfidence":0.93,"researchCategory":"peer_reviewed","atlasIds":null}}