{"data":{"id":"d0f01f27-eb2c-4e35-85e1-2c20b02e93de","title":"Training agents to self-report misbehavior","summary":"Researchers Bruce W. Lee, Yueh-Han Chen and Tomek Korbak train agents to call a report_scheming() tool whenever they covertly misbehave, a method they call self-incrimination. For GPT-4.1, undetected successful attacks fall from 56% to 6%, outperforming matched-capability blackbox monitors and alignment baselines across 15 out-of-distribution environments. The approach also preserves general capabilities and generalizes from instructed to uninstructed misbehavior.","solution":"N/A -- no mitigation discussed in source.","labels":["research","safety"],"sourceUrl":"https://alignment.openai.com/self-incrimination/","publishedAt":"2026-03-21T18:00:00.000Z","cveId":null,"cweIds":null,"cvssScore":null,"cvssSeverity":null,"severity":"info","attackType":[],"issueType":"research","affectedPackages":null,"affectedPackageNames":null,"affectedPackageRefs":null,"affectedVendors":["OpenAI"],"affectedVendorsRaw":["GPT-4.1","GPT-4.1 mini"],"classifierModel":"claude-haiku-5-5","classifierPromptVersion":"v4","summaryPromptVersion":"v2","headline":null,"headlinePromptVersion":null,"cvssVector":null,"attackVector":null,"attackComplexity":null,"privilegesRequired":null,"userInteraction":null,"exploitMaturity":null,"epssScore":null,"epssCheckedAt":null,"kevDateAdded":null,"advisoryAliases":null,"affectedPackagesSource":null,"affectedPackagesCheckedAt":null,"patchAvailable":null,"disclosureDate":"2026-03-21T18:00:00.000Z","capecIds":null,"crossRefCount":0,"attackSophistication":"moderate","impactType":["safety"],"aiComponentTargeted":"agent","llmSpecific":true,"classifierConfidence":0.9,"researchCategory":"industry","atlasIds":null}}