{"data":{"id":"c7b32c66-b96c-4b7f-83b3-756a4498ad13","title":"A Survey of Direct Preference Optimization: Datasets, Theories, Variants, and Applications","summary":"This paper surveys Direct Preference Optimization (DPO), a method for aligning large language models (AI systems trained on massive amounts of text) with human values and preferences without using reinforcement learning (a training approach that rewards desired behaviors). The survey reviews the theoretical foundations, different versions of DPO, available datasets of human preferences, and real-world applications, while also identifying current limitations and suggesting directions for future research.","solution":"N/A -- no mitigation discussed in source.","labels":["research"],"sourceUrl":"http://ieeexplore.ieee.org/document/11568678","publishedAt":"2026-06-16T13:16:21.000Z","cveId":null,"cweIds":null,"cvssScore":null,"cvssSeverity":null,"severity":"info","attackType":[],"issueType":"research","affectedPackages":null,"affectedVendors":[],"affectedVendorsRaw":[],"classifierModel":"claude-haiku-4-5-20251001","classifierPromptVersion":"v3","cvssVector":null,"attackVector":null,"attackComplexity":null,"privilegesRequired":null,"userInteraction":null,"exploitMaturity":null,"epssScore":null,"patchAvailable":null,"disclosureDate":"2026-06-16T13:16:21.000Z","capecIds":null,"crossRefCount":0,"attackSophistication":"moderate","impactType":null,"aiComponentTargeted":"model","llmSpecific":true,"classifierConfidence":0.95,"researchCategory":"peer_reviewed","atlasIds":null}}