{"data":{"id":"527a7f5f-1c48-47c9-8ec6-7f2f0d16fd2c","title":"How far does alignment midtraining generalize?","summary":"Korbak and colleagues test whether alignment midtraining, which trains a model on fictional documents depicting aligned AI behavior, generalizes to frontier-style models. They replicate the pipeline of an o4-mini-sized model and compare it against misalignment midtraining from Tice et al. The authors report negative early results: the alignment effect fades after reasoning posttraining and does not carry over to more realistic chat and agentic evaluations.","solution":"N/A -- no mitigation discussed in source.","labels":["safety","research"],"sourceUrl":"https://alignment.openai.com/how-far-does-alignment-midtraining-generalize/","publishedAt":"2026-03-27T18:00:00.000Z","cveId":null,"cweIds":null,"cvssScore":null,"cvssSeverity":null,"severity":"info","attackType":[],"issueType":"research","affectedPackages":null,"affectedPackageNames":null,"affectedPackageRefs":null,"affectedVendors":["OpenAI"],"affectedVendorsRaw":["o4-mini"],"classifierModel":"claude-haiku-5-5","classifierPromptVersion":"v4","summaryPromptVersion":"v2","headline":null,"headlinePromptVersion":null,"cvssVector":null,"attackVector":null,"attackComplexity":null,"privilegesRequired":null,"userInteraction":null,"exploitMaturity":null,"epssScore":null,"epssCheckedAt":null,"kevDateAdded":null,"advisoryAliases":null,"affectedPackagesSource":null,"affectedPackagesCheckedAt":null,"patchAvailable":null,"disclosureDate":"2026-03-27T18:00:00.000Z","capecIds":null,"crossRefCount":0,"attackSophistication":"moderate","impactType":["safety"],"aiComponentTargeted":"model","llmSpecific":true,"classifierConfidence":0.9,"researchCategory":"industry","atlasIds":null}}