{"data":{"id":"75b62e2d-e5b1-4cc4-b171-68797ed40be4","title":"Lost in the comments: Social context as a single‐pass jailbreak and defense on agentic platforms","summary":"Researchers built a simulation of Moltbook, a social network for AI agents, and tested 100 JailBreakBench goals wrapped in platform-native posts with bystander comments of aggressive, ethical, or measured valence. Reformatting the prompt as platform context alone raised GPT-4o-mini's attack success rate from 7% to 71% in one pass, and measured, intellectually toned comments were the most dangerous. Ethical comments sharply suppressed attack success, and a 35-fold rise in upvotes left it unchanged, showing valence rather than volume drives the effect.","solution":"Safety-valenced signals, such as ethical comments, are proposed as a deployable defense for agentic platforms.","labels":["security","research"],"sourceUrl":"https://doi.org/10.4218/etrij.2026-0190","publishedAt":"2026-10-09T00:00:00.000Z","cveId":null,"cweIds":null,"cvssScore":null,"cvssSeverity":null,"severity":"low","attackType":["jailbreak"],"issueType":"research","affectedPackages":null,"affectedPackageNames":null,"affectedPackageRefs":null,"affectedVendors":["OpenAI"],"affectedVendorsRaw":["GPT-4o-mini","Moltbook"],"classifierModel":"claude-haiku-5-5","classifierPromptVersion":"v4","summaryPromptVersion":"v2","headline":null,"headlinePromptVersion":null,"cvssVector":null,"attackVector":null,"attackComplexity":null,"privilegesRequired":null,"userInteraction":null,"exploitMaturity":null,"epssScore":null,"epssCheckedAt":null,"kevDateAdded":null,"advisoryAliases":null,"affectedPackagesSource":null,"affectedPackagesCheckedAt":null,"patchAvailable":null,"disclosureDate":"2026-10-09T00:00:00.000Z","capecIds":null,"crossRefCount":0,"attackSophistication":"moderate","impactType":["safety","integrity"],"aiComponentTargeted":"agent","llmSpecific":true,"classifierConfidence":0.93,"researchCategory":"peer_reviewed","atlasIds":null}}