{"data":{"id":"f393dd49-54bf-496d-a043-fee6d14cc99b","title":"LLMs Cannot Reliably Judge (Yet?): A Comprehensive Assessment on the Robustness of LLM-as-a-Judge","summary":"RobustJudge is an automated, modular framework that systematically tests how robust LLM-as-a-Judge systems are across datasets, prompt templates, judge models, attacks and defenses. The study covers 15 attack methods and 8 defense strategies across 13 models, and finds that LLM-based judges stay susceptible under both pointwise and pairwise protocols. Robustness is highly sensitive to prompt-template and judge-model choice, and optimization-based attacks with long suffixes can substantially inflate scores from both PAI-Judge variants.","solution":"N/A -- no mitigation discussed in source.","labels":["security","research"],"sourceUrl":"http://ieeexplore.ieee.org/document/11711198","publishedAt":"2026-09-25T05:06:31.000Z","cveId":null,"cweIds":null,"cvssScore":null,"cvssSeverity":null,"severity":"info","attackType":["model_evasion"],"issueType":"research","affectedPackages":null,"affectedPackageNames":null,"affectedVendors":[],"affectedVendorsRaw":[],"classifierModel":"claude-haiku-5-5","classifierPromptVersion":"v4","summaryPromptVersion":"v2","headline":null,"headlinePromptVersion":null,"cvssVector":null,"attackVector":null,"attackComplexity":null,"privilegesRequired":null,"userInteraction":null,"exploitMaturity":null,"epssScore":null,"epssCheckedAt":null,"kevDateAdded":null,"advisoryAliases":null,"affectedPackagesSource":null,"affectedPackagesCheckedAt":null,"patchAvailable":null,"disclosureDate":"2026-09-25T05:06:31.000Z","capecIds":null,"crossRefCount":0,"attackSophistication":"advanced","impactType":["integrity"],"aiComponentTargeted":"model","llmSpecific":true,"classifierConfidence":0.95,"researchCategory":"peer_reviewed","atlasIds":null}}