{"data":{"id":"754a4ee5-f297-478b-851a-6178a00df8f9","title":"Investigating the consequences of accidentally grading CoT during RL","summary":"OpenAI's researchers report that an automated system found limited accidental Chain-of-Thought (CoT) grading during RL training of some released models, including GPT-5.4 Thinking, GPT-5.1 Instant through GPT-5.4 Instant, GPT-5.3 mini, and GPT-5.4 mini, despite the company's policy against it. Their analysis found no clear reduction in CoT monitorability, though they cannot rule out harder-to-measure effects.","solution":"Fixed the affected reward pathways, expanding the automated detection system, and strengthened internal guidance and processes.","labels":["safety","research"],"sourceUrl":"https://alignment.openai.com/accidental-cot-grading/","publishedAt":"2026-05-06T23:00:00.000Z","cveId":null,"cweIds":null,"cvssScore":null,"cvssSeverity":null,"severity":"info","attackType":[],"issueType":"research","affectedPackages":null,"affectedPackageNames":null,"affectedPackageRefs":null,"affectedVendors":["OpenAI"],"affectedVendorsRaw":["GPT-5.4 Thinking","GPT-5.1 Instant","GPT-5.4 Instant","GPT-5.3 mini","GPT-5.4 mini","GPT-5.5"],"classifierModel":"claude-haiku-5-5","classifierPromptVersion":"v4","summaryPromptVersion":"v2","headline":null,"headlinePromptVersion":null,"cvssVector":null,"attackVector":null,"attackComplexity":null,"privilegesRequired":null,"userInteraction":null,"exploitMaturity":null,"epssScore":null,"epssCheckedAt":null,"kevDateAdded":null,"advisoryAliases":null,"affectedPackagesSource":null,"affectedPackagesCheckedAt":null,"patchAvailable":null,"disclosureDate":"2026-05-06T23:00:00.000Z","capecIds":null,"crossRefCount":0,"attackSophistication":"moderate","impactType":["integrity"],"aiComponentTargeted":"model","llmSpecific":true,"classifierConfidence":0.85,"researchCategory":"industry","atlasIds":null}}