{"data":{"id":"c181e34c-a7d4-4ce9-a2b9-dc75bafc1a39","title":"Introducing Model Spec Evals","summary":"OpenAI released Model Spec Evals, a suite that measures how well models follow the OpenAI Model Spec, along with 596 evaluation prompts and open-source evaluation code. Compliance rates were 72% for GPT-4o, 80% for OpenAI o3, 82% for GPT-5 Instant, 89% for GPT-5 Thinking, 84% for GPT-5.3 Instant, and 87% for GPT-5.4 Thinking. The evaluations cover only text-only interactions, and the prompt collection is small relative to the Spec's scope.","solution":"N/A -- no mitigation discussed in source.","labels":["safety","research"],"sourceUrl":"https://alignment.openai.com/model-spec-evals/","publishedAt":"2026-03-25T17:00:00.000Z","cveId":null,"cweIds":null,"cvssScore":null,"cvssSeverity":null,"severity":"info","attackType":[],"issueType":"research","affectedPackages":null,"affectedPackageNames":null,"affectedPackageRefs":null,"affectedVendors":["OpenAI"],"affectedVendorsRaw":["OpenAI Model Spec","GPT-4o","OpenAI o3","GPT-5 Instant","GPT-5 Thinking","GPT-5.3 Instant","GPT-5.4 Thinking","ChatGPT"],"classifierModel":"claude-haiku-5-5","classifierPromptVersion":"v4","summaryPromptVersion":"v2","headline":null,"headlinePromptVersion":null,"cvssVector":null,"attackVector":null,"attackComplexity":null,"privilegesRequired":null,"userInteraction":null,"exploitMaturity":null,"epssScore":null,"epssCheckedAt":null,"kevDateAdded":null,"advisoryAliases":null,"affectedPackagesSource":null,"affectedPackagesCheckedAt":null,"patchAvailable":null,"disclosureDate":"2026-03-25T17:00:00.000Z","capecIds":null,"crossRefCount":0,"attackSophistication":"moderate","impactType":null,"aiComponentTargeted":null,"llmSpecific":true,"classifierConfidence":0.8,"researchCategory":"industry","atlasIds":null}}