{"data":{"id":"64d7787c-6397-4802-9733-f4f8f0d2a858","title":"Sidestepping Evaluation Awareness and Anticipating Misalignment with Production Evaluations","summary":"Marcus Williams, Cameron Raymond and Micah Carroll describe a pipeline that uses de-identified ChatGPT production traffic to build realistic alignment evaluations. The method resamples the final model response from each conversation and labels the new responses with LLM monitors, either to discover unknown misaligned behaviors or to estimate how often known ones occur. The authors note the pipeline depends on a monitor's ability to detect undesirable behaviors and cannot guarantee catching all of them.","solution":"N/A -- no mitigation discussed in source.","labels":["research","safety"],"sourceUrl":"https://alignment.openai.com/prod-evals/","publishedAt":"2025-12-18T19:00:00.000Z","cveId":null,"cweIds":null,"cvssScore":null,"cvssSeverity":null,"severity":"info","attackType":[],"issueType":"research","affectedPackages":null,"affectedPackageNames":null,"affectedPackageRefs":null,"affectedVendors":["OpenAI"],"affectedVendorsRaw":["OpenAI","ChatGPT"],"classifierModel":"claude-haiku-5-5","classifierPromptVersion":"v4","summaryPromptVersion":"v2","headline":null,"headlinePromptVersion":null,"cvssVector":null,"attackVector":null,"attackComplexity":null,"privilegesRequired":null,"userInteraction":null,"exploitMaturity":null,"epssScore":null,"epssCheckedAt":null,"kevDateAdded":null,"advisoryAliases":null,"affectedPackagesSource":null,"affectedPackagesCheckedAt":null,"patchAvailable":null,"disclosureDate":"2025-12-18T19:00:00.000Z","capecIds":null,"crossRefCount":0,"attackSophistication":"moderate","impactType":["safety"],"aiComponentTargeted":"model","llmSpecific":true,"classifierConfidence":0.93,"researchCategory":"industry","atlasIds":null}}