{"data":{"id":"77e9e9d9-7198-4ce0-8939-3e12e73b0d21","title":"Can public chat data predict real-world AI misalignments?","summary":"Researchers ask whether external groups can evaluate frontier language models by substituting WildChat, a public dataset of about 1 million conversations collected between April 2023 and May 2024, for private production data in their Deployment Simulation technique. They report that WildChat-based predictions of real-world failure rates are surprisingly accurate, typically within roughly 3x error for GPT-5.1, 5.2 and 5.4, despite a 2-3 year data gap. Predictive performance degrades most for more technical and agentic forms of misalignment.","solution":"N/A -- no mitigation discussed in source.","labels":["research","safety"],"sourceUrl":"https://alignment.openai.com/validating-public-evals/","publishedAt":"2026-06-16T18:00:00.000Z","cveId":null,"cweIds":null,"cvssScore":null,"cvssSeverity":null,"severity":"info","attackType":[],"issueType":"research","affectedPackages":null,"affectedPackageNames":null,"affectedPackageRefs":null,"affectedVendors":["OpenAI"],"affectedVendorsRaw":["GPT-5.1","GPT-5.2","GPT-5.4","ChatGPT-3.5","GPT-4","WildChat"],"classifierModel":"claude-haiku-5-5","classifierPromptVersion":"v4","summaryPromptVersion":"v2","headline":null,"headlinePromptVersion":null,"cvssVector":null,"attackVector":null,"attackComplexity":null,"privilegesRequired":null,"userInteraction":null,"exploitMaturity":null,"epssScore":null,"epssCheckedAt":null,"kevDateAdded":null,"advisoryAliases":null,"affectedPackagesSource":null,"affectedPackagesCheckedAt":null,"patchAvailable":null,"disclosureDate":"2026-06-16T18:00:00.000Z","capecIds":null,"crossRefCount":0,"attackSophistication":"moderate","impactType":null,"aiComponentTargeted":"model","llmSpecific":true,"classifierConfidence":0.85,"researchCategory":"industry","atlasIds":null}}