{"data":{"id":"11d27316-f13b-49d4-ae4c-53786a60ff70","title":"A2Net: Affiliation Alignment Networks for Whole-Body Pose Estimation With Vision--Language Models","summary":"This paper introduces A2Net, a system for whole-body pose estimation (predicting the locations of keypoints on a person's face, body, hands, and feet from an image) that combines vision and language models to solve two problems: scale variation (different body parts appearing at different sizes) and semantic ambiguity in small-scale parts (difficulty identifying what small features represent). The approach uses text features alongside image features because text is not affected by image scaling issues, then aligns them using optimal transport (a mathematical method for matching distributions) to create a unified visual-language representation that improves keypoint localization accuracy.","solution":"N/A -- no mitigation discussed in source.","labels":["research"],"sourceUrl":"http://ieeexplore.ieee.org/document/11373586","publishedAt":"2026-02-06T13:32:30.000Z","cveId":null,"cweIds":null,"cvssScore":null,"cvssSeverity":null,"severity":"info","attackType":[],"issueType":"research","affectedPackages":null,"affectedVendors":[],"affectedVendorsRaw":[],"classifierModel":"claude-haiku-4-5-20251001","classifierPromptVersion":"v3","cvssVector":null,"attackVector":null,"attackComplexity":null,"privilegesRequired":null,"userInteraction":null,"exploitMaturity":null,"epssScore":null,"patchAvailable":null,"disclosureDate":"2026-02-06T13:32:30.000Z","capecIds":null,"crossRefCount":0,"attackSophistication":"moderate","impactType":null,"aiComponentTargeted":"model","llmSpecific":false,"classifierConfidence":0.75,"researchCategory":"peer_reviewed","atlasIds":null}}