{"data":{"id":"cbd8a03e-6056-48b2-9d9a-623de8cb04bb","title":"Why we are excited about confessions","summary":"Boaz Barak, Gabriel Wu, Jeremy Chen and Manas Joglekar publish a follow-up to their confessions paper, giving deeper analysis of how training affects confessions and preliminary comparisons to chain-of-thought monitoring. The approach trains a second model output, a confession, rewarded solely for honesty about misbehavior in the main task. The authors hypothesize that honest confessions are the path of least resistance because confessing is easier than sustaining an elaborate lie.","solution":"N/A -- no mitigation discussed in source.","labels":["safety","research"],"sourceUrl":"https://alignment.openai.com/confessions/","publishedAt":"2026-01-12T19:00:00.000Z","cveId":null,"cweIds":null,"cvssScore":null,"cvssSeverity":null,"severity":"info","attackType":[],"issueType":"research","affectedPackages":null,"affectedPackageNames":null,"affectedPackageRefs":null,"affectedVendors":["OpenAI"],"affectedVendorsRaw":["OpenAI"],"classifierModel":"claude-haiku-5-5","classifierPromptVersion":"v4","summaryPromptVersion":"v2","headline":null,"headlinePromptVersion":null,"cvssVector":null,"attackVector":null,"attackComplexity":null,"privilegesRequired":null,"userInteraction":null,"exploitMaturity":null,"epssScore":null,"epssCheckedAt":null,"kevDateAdded":null,"advisoryAliases":null,"affectedPackagesSource":null,"affectedPackagesCheckedAt":null,"patchAvailable":null,"disclosureDate":"2026-01-12T19:00:00.000Z","capecIds":null,"crossRefCount":0,"attackSophistication":"moderate","impactType":["integrity"],"aiComponentTargeted":"model","llmSpecific":true,"classifierConfidence":0.9,"researchCategory":"industry","atlasIds":null}}