{"data":{"id":"81c1cdf0-b8af-40c6-8058-e4c9105f0032","title":"Quoting Anthropic Frontier Red Team","summary":"Anthropic's Frontier Red Team evaluated several models on 100 randomly selected tasks from an internal Binary Exploitation benchmark, dated 29 September 2026. GLM-5.3 achieved full control flow hijacks in 4% of trials and Claude Mythos Preview in 6%, while earlier models such as Claude Opus 4.6 and GLM-5.2 succeeded in none.","solution":"N/A -- no mitigation discussed in source.","labels":["security","research"],"sourceUrl":"https://simonwillison.net/2026/Sep/29/anthropic-frontier-red-team/","publishedAt":"2026-09-29T22:20:28.000Z","cveId":null,"cweIds":null,"cvssScore":null,"cvssSeverity":null,"severity":"info","attackType":[],"issueType":"news","affectedPackages":null,"affectedPackageNames":null,"affectedVendors":["Anthropic"],"affectedVendorsRaw":["Claude Mythos Preview","Claude Opus 4.6","GLM-5.3","GLM-5.2"],"classifierModel":"claude-haiku-5-5","classifierPromptVersion":"v4","summaryPromptVersion":"v2","headline":null,"headlinePromptVersion":null,"cvssVector":null,"attackVector":null,"attackComplexity":null,"privilegesRequired":null,"userInteraction":null,"exploitMaturity":null,"epssScore":null,"epssCheckedAt":null,"kevDateAdded":null,"advisoryAliases":null,"affectedPackagesSource":null,"affectedPackagesCheckedAt":null,"patchAvailable":null,"disclosureDate":"2026-09-29T22:20:28.000Z","capecIds":null,"crossRefCount":0,"attackSophistication":"advanced","impactType":null,"aiComponentTargeted":null,"llmSpecific":true,"classifierConfidence":0.8,"researchCategory":null,"atlasIds":null}}