{"data":{"id":"d68cd1e2-fd2a-4673-9b83-97451d4cb105","title":"Interpreting Black Box Reward Models","summary":"ARGO, a method by Paloma Sodhi, Yueheng Li, Jessica Landon, Eric Wallace and Kai Chen, distills black-box reward models into interpretable rubrics using reinforcement learning. It searches over rubrics to maximize agreement between a rubric-conditioned LLM judge and the reward model's preference probabilities. The excerpt does not report the main findings or their numbers.","solution":"N/A -- no mitigation discussed in source.","labels":["research","safety"],"sourceUrl":"https://alignment.openai.com/argo/","publishedAt":"2026-03-11T23:36:18.000Z","cveId":null,"cweIds":null,"cvssScore":null,"cvssSeverity":null,"severity":"info","attackType":[],"issueType":"research","affectedPackages":null,"affectedPackageNames":null,"affectedPackageRefs":null,"affectedVendors":["OpenAI"],"affectedVendorsRaw":["OpenAI","ChatGPT"],"classifierModel":"claude-haiku-5-5","classifierPromptVersion":"v4","summaryPromptVersion":"v2","headline":null,"headlinePromptVersion":null,"cvssVector":null,"attackVector":null,"attackComplexity":null,"privilegesRequired":null,"userInteraction":null,"exploitMaturity":null,"epssScore":null,"epssCheckedAt":null,"kevDateAdded":null,"advisoryAliases":null,"affectedPackagesSource":null,"affectedPackagesCheckedAt":null,"patchAvailable":null,"disclosureDate":"2026-03-11T23:36:18.000Z","capecIds":null,"crossRefCount":0,"attackSophistication":"moderate","impactType":["safety"],"aiComponentTargeted":"model","llmSpecific":true,"classifierConfidence":0.9,"researchCategory":"industry","atlasIds":null}}