{"data":{"id":"c9df9269-27ad-461c-a1a3-d28468f94ff1","title":"Evaluating Safety Embedding Prefiltering for Analyzing Millions of LLM Agent Social Network Messages for Security and Safety Harms","summary":"The study examines whether lightweight embedding-based prefilters can cut the cost of screening agentic social network content for security and safety harms. Researchers annotated 10,000 Moltbook posts and comments with a frontier LLM judge, finding 9.2% unsafe at severity 3 or above. At 0.80 recall, prefiltering reduced the projected cost of scanning 787,226 messages by 49.9% to 65.6%, depending on the encoder or trained classifier, though jailbreak content remained the hardest category to retrieve.","solution":"N/A -- no mitigation discussed in source.","labels":["security","research"],"sourceUrl":"https://doi.org/10.3390/ai7100409","publishedAt":"2026-10-06T00:00:00.000Z","cveId":null,"cweIds":null,"cvssScore":null,"cvssSeverity":null,"severity":"low","attackType":["prompt_injection","jailbreak"],"issueType":"research","affectedPackages":null,"affectedPackageNames":null,"affectedPackageRefs":null,"affectedVendors":[],"affectedVendorsRaw":["Moltbook","MiniLM-L12-v2","BGE-M3"],"classifierModel":"claude-haiku-5-5","classifierPromptVersion":"v4","summaryPromptVersion":"v2","headline":null,"headlinePromptVersion":null,"cvssVector":null,"attackVector":null,"attackComplexity":null,"privilegesRequired":null,"userInteraction":null,"exploitMaturity":null,"epssScore":null,"epssCheckedAt":null,"kevDateAdded":null,"advisoryAliases":null,"affectedPackagesSource":null,"affectedPackagesCheckedAt":null,"patchAvailable":null,"disclosureDate":"2026-10-06T00:00:00.000Z","capecIds":null,"crossRefCount":0,"attackSophistication":"moderate","impactType":["integrity","safety"],"aiComponentTargeted":"agent","llmSpecific":true,"classifierConfidence":0.9,"researchCategory":"peer_reviewed","atlasIds":null}}