[
  {
    "schema_version": 1,
    "slug": "embodied-brains",
    "canonical_url": "https://akira-l.github.io/publications/embodied-brains/",
    "language_urls": {
      "en": "https://akira-l.github.io/publications/embodied-brains/",
      "zh-CN": "https://akira-l.github.io/zh/publications/embodied-brains/"
    },
    "title": "From World Action Models to Embodied Brains: A Roadmap for Open-World Physical Intelligence",
    "short_title": "Embodied Brains Roadmap",
    "authors": [
      "Yuanzhi Liang",
      "Xufeng Zhan",
      "Haibin Huang",
      "Chi Zhang",
      "Xuelong Li"
    ],
    "publication": {
      "kind": "preprint",
      "venue": "arXiv",
      "citation_container_title": "arXiv",
      "status": "preprint",
      "year": 2026,
      "publication_date": "2026-07-13"
    },
    "identifiers": {
      "arxiv": "2607.11689",
      "arxiv_primary_class": "cs.RO"
    },
    "arxiv_dates": {
      "first_posted": "2026-07-13",
      "last_revised": "2026-07-13"
    },
    "keywords": [
      "world action models",
      "embodied intelligence",
      "physical intelligence",
      "world models",
      "robotics"
    ],
    "official_abstract": "Artificial general intelligence ultimately requires agents that can reason and act in the physical world. Action models, vision-language-action policies, and world models have advanced this goal, while World Action Models (WAMs) are particularly promising because they connect candidate interventions with predicted consequences. However, progress remains fragmented: models use incompatible action spaces and prediction targets, datasets and tasks follow different conventions, and runtime systems expose limited interfaces for reuse and evaluation. We review the evolution toward WAMs and organize these limitations into three coupled gaps: model roles and representations, objectives and standardization, and system composition. Building on this analysis, we propose a co-evolution roadmap for physical intelligence centered on the embodied brain, a long-term model target for integrating multimodal context, comparing candidate interventions, and issuing state-transition or capability requests rather than direct actuator commands. WAMs provide promising prototypes for its predictive functions, while a physical harness grounds model outputs through tools, controllers, verification, and trace logging. Shared contracts align heterogeneous models, data, tasks, and embodiments, and closed-loop post-training converts verified interaction into reusable experience. Together, these components define a modular physical-intelligence stack for adaptive and self-improving embodied agents.",
    "official_abstract_rights": "Excluded from the site's CC BY 4.0 license; original paper rights apply.",
    "primary_source": {
      "label": "arXiv record",
      "url": "https://arxiv.org/abs/2607.11689",
      "version": "arXiv:2607.11689v1",
      "method_locator": "Abstract; roadmap sections on embodied brains and physical harnesses",
      "evidence_locator": "Abstract; paper synthesis and roadmap discussion"
    },
    "source_checked": "2026-07-31",
    "verification_status": "author-verified",
    "author_verified_on": "2026-07-31",
    "commentary": {
      "license": "https://creativecommons.org/licenses/by/4.0/",
      "en": {
        "summary": "This roadmap places World Action Models inside a broader physical-intelligence stack: an embodied brain compares possible interventions, a physical harness grounds its requests through tools and controllers, shared contracts connect heterogeneous components, and verified interaction becomes post-training experience.",
        "problem": "Research on action models, vision-language-action policies, and world models is advancing, but incompatible representations, objectives, datasets, tasks, and runtime interfaces make the resulting systems difficult to compose, evaluate, and improve as a whole.",
        "contributions": [
          "Organizes the field's limitations into coupled gaps in model roles and representations, objectives and standardization, and system composition.",
          "Defines the embodied brain as a long-term model target that reasons over multimodal context and requests state transitions or capabilities instead of directly commanding actuators.",
          "Proposes physical harnesses, shared contracts, and closed-loop post-training as the system mechanisms that ground, connect, verify, and reuse model behavior."
        ],
        "evidence": "The paper is a review and roadmap. Its support is a structured synthesis of prior action-model, VLA, and world-model research and a systems argument for the proposed stack; it does not present a newly deployed embodied system or a standalone empirical benchmark.",
        "limitations": "The architecture is a forward-looking research agenda. Individual contracts, verification mechanisms, harness implementations, and closed-loop training procedures still require concrete specifications and empirical validation across embodiments.",
        "positioning": "Use this work when discussing how predictive world/action models can become reusable components of open-world embodied systems. Its distinctive contribution is the co-design of model roles, standardized interfaces, runtime grounding, and learning from verified interaction.",
        "citation_ready": "Liang et al. present a roadmap from World Action Models to embodied brains, arguing that predictive models should be integrated with physical harnesses, shared contracts, and closed-loop post-training to support modular open-world physical intelligence."
      },
      "zh-CN": {
        "summary": "这篇路线图把 World Action Models 放入更完整的物理智能系统：embodied brain 比较候选干预，physical harness 通过工具和控制器将请求落地，共享 contracts 连接异构组件，经过验证的交互再转化为后训练经验。",
        "problem": "Action model、VLA policy 与 world model 虽然持续发展，但它们在表示、目标、数据集、任务约定和运行时接口上彼此割裂，难以被组合、复用、统一评测并形成持续改进的系统。",
        "contributions": [
          "把领域瓶颈归纳为模型角色与表示、目标与标准化、系统组合三个相互耦合的缺口。",
          "提出 embodied brain 这一长期目标：结合多模态上下文比较候选干预，输出状态转移或能力请求，而不是直接下发执行器指令。",
          "以 physical harness、共享 contracts 和闭环后训练作为模型落地、组件连接、行为验证与经验复用的系统机制。"
        ],
        "evidence": "本文属于综述与路线图，证据主要来自对 action model、VLA 和 world model 文献的结构化梳理以及系统设计论证；它并未声称已经实现一个完整部署的 embodied brain 或新的统一基准。",
        "limitations": "该方案仍是面向未来的研究议程。不同 embodiment 下的接口规范、验证机制、harness 实现与闭环训练流程都需要进一步工程化和实证检验。",
        "positioning": "在讨论预测型 world/action model 如何成为开放世界具身系统的可复用组件时可以引用这项工作。其差异点是把模型角色、标准接口、运行时落地和验证交互驱动的学习放在同一条演进路线中。",
        "citation_ready": "Liang 等提出了从 World Action Models 走向 embodied brains 的路线图，主张将预测模型与 physical harness、共享 contracts 和闭环后训练结合，以构建模块化的开放世界物理智能系统。"
      }
    },
    "citation": {
      "id": "liang2026embodiedbrains",
      "type": "article",
      "title": "From World Action Models to Embodied Brains: A Roadmap for Open-World Physical Intelligence",
      "author": [
        {
          "given": "Yuanzhi",
          "family": "Liang"
        },
        {
          "given": "Xufeng",
          "family": "Zhan"
        },
        {
          "given": "Haibin",
          "family": "Huang"
        },
        {
          "given": "Chi",
          "family": "Zhang"
        },
        {
          "given": "Xuelong",
          "family": "Li"
        }
      ],
      "container-title": "arXiv",
      "issued": {
        "date-parts": [
          [
            2026,
            7,
            13
          ]
        ]
      },
      "URL": "https://arxiv.org/abs/2607.11689",
      "abstract": "Artificial general intelligence ultimately requires agents that can reason and act in the physical world. Action models, vision-language-action policies, and world models have advanced this goal, while World Action Models (WAMs) are particularly promising because they connect candidate interventions with predicted consequences. However, progress remains fragmented: models use incompatible action spaces and prediction targets, datasets and tasks follow different conventions, and runtime systems expose limited interfaces for reuse and evaluation. We review the evolution toward WAMs and organize these limitations into three coupled gaps: model roles and representations, objectives and standardization, and system composition. Building on this analysis, we propose a co-evolution roadmap for physical intelligence centered on the embodied brain, a long-term model target for integrating multimodal context, comparing candidate interventions, and issuing state-transition or capability requests rather than direct actuator commands. WAMs provide promising prototypes for its predictive functions, while a physical harness grounds model outputs through tools, controllers, verification, and trace logging. Shared contracts align heterogeneous models, data, tasks, and embodiments, and closed-loop post-training converts verified interaction into reusable experience. Together, these components define a modular physical-intelligence stack for adaptive and self-improving embodied agents.",
      "keyword": "world action models, embodied intelligence, physical intelligence, world models, robotics",
      "archive": "arXiv",
      "archive_location": "2607.11689",
      "genre": "Preprint"
    },
    "resources": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2607.11689"
      },
      {
        "label": "HTML",
        "url": "https://arxiv.org/html/2607.11689v1"
      },
      {
        "label": "PDF",
        "url": "https://arxiv.org/pdf/2607.11689"
      }
    ]
  },
  {
    "schema_version": 1,
    "slug": "teleboost",
    "canonical_url": "https://akira-l.github.io/publications/teleboost/",
    "language_urls": {
      "en": "https://akira-l.github.io/publications/teleboost/",
      "zh-CN": "https://akira-l.github.io/zh/publications/teleboost/"
    },
    "title": "TeleBoost: A Systematic Alignment Framework for High-Fidelity, Controllable, and Robust Video Generation",
    "short_title": "TeleBoost",
    "authors": [
      "Yuanzhi Liang",
      "Xuan'er Wu",
      "Yirui Liu",
      "Yijie Fang",
      "Yizhen Fan",
      "Ke Hao",
      "Rui Li",
      "Ruiying Liu",
      "Ziqi Ni",
      "Peng Yu",
      "Yanbo Wang",
      "Haibin Huang",
      "Qizhen Weng",
      "Chi Zhang",
      "Xuelong Li"
    ],
    "publication": {
      "kind": "preprint",
      "venue": "arXiv",
      "citation_container_title": "arXiv",
      "status": "preprint",
      "year": 2026,
      "publication_date": "2026-02-07"
    },
    "identifiers": {
      "arxiv": "2602.07595",
      "arxiv_primary_class": "cs.CV"
    },
    "arxiv_dates": {
      "first_posted": "2026-02-07",
      "last_revised": "2026-02-07"
    },
    "keywords": [
      "video generation",
      "post-training",
      "alignment",
      "reinforcement learning",
      "preference optimization"
    ],
    "official_abstract": "Post-training is the decisive step for converting a pretrained video generator into a production-oriented model that is instruction-following, controllable, and robust over long temporal horizons. This report presents a systematical post-training framework that organizes supervised policy shaping, reward-driven reinforcement learning, and preference-based refinement into a single stability-constrained optimization stack. The framework is designed around practical video-generation constraints, including high rollout cost, temporally compounding failure modes, and feedback that is heterogeneous, uncertain, and often weakly discriminative. By treating optimization as a staged, diagnostic-driven process rather than a collection of isolated tricks, the report summarizes a cohesive recipe for improving perceptual fidelity, temporal coherence, and prompt adherence while preserving the controllability established at initialization. The resulting framework provides a clear blueprint for building scalable post-training pipelines that remain stable, extensible, and effective in real-world deployment settings.",
    "official_abstract_rights": "Excluded from the site's CC BY 4.0 license; original paper rights apply.",
    "primary_source": {
      "label": "arXiv record",
      "url": "https://arxiv.org/abs/2602.07595",
      "version": "arXiv:2602.07595v1",
      "method_locator": "Abstract; framework sections on supervised shaping, reward-driven RL, and preference refinement",
      "evidence_locator": "Abstract; report's diagnostic and deployment-oriented analysis"
    },
    "source_checked": "2026-07-31",
    "verification_status": "author-verified",
    "author_verified_on": "2026-07-31",
    "commentary": {
      "license": "https://creativecommons.org/licenses/by/4.0/",
      "en": {
        "summary": "TeleBoost treats video post-training as a staged, stability-constrained system that connects supervised policy shaping, reward-driven reinforcement learning, and preference-based refinement while diagnosing costly rollouts, compounding temporal failures, and uncertain feedback.",
        "problem": "A pretrained video generator is not automatically instruction-following, controllable, temporally robust, or deployment-ready. Video rollouts are expensive, errors compound over time, and heterogeneous feedback can be uncertain or weakly discriminative.",
        "contributions": [
          "Organizes supervised policy shaping, reward-driven RL, and preference refinement into one staged optimization stack.",
          "Frames diagnostics and stability constraints as first-class components of video-model post-training.",
          "Provides a deployment-oriented blueprint for improving fidelity, temporal coherence, and prompt adherence while preserving initial controllability."
        ],
        "evidence": "TeleBoost is a systematic framework and report rather than a single isolated algorithm. Its claims should be read as a combined post-training recipe supported by the report's analyses and experiments, not as evidence that any one stage alone produces all reported properties.",
        "limitations": "The framework requires expensive video rollouts, multiple feedback sources, diagnostic infrastructure, and careful stage-specific tuning. Reproduction and comparison should preserve the full training stack and initialization assumptions.",
        "positioning": "TeleBoost is suitable for related work on production-oriented video-generation alignment and post-training systems. It emphasizes orchestration and stability across SFT-, RL-, and preference-based stages rather than proposing only one optimizer.",
        "citation_ready": "Liang et al. present TeleBoost, a stability-constrained video-generation post-training framework that stages supervised policy shaping, reward-driven reinforcement learning, and preference refinement to improve fidelity, controllability, temporal coherence, and prompt adherence."
      },
      "zh-CN": {
        "summary": "TeleBoost 把视频后训练视为分阶段、受稳定性约束的系统：串联 supervised policy shaping、reward-driven RL 与 preference refinement，并针对高昂 rollout、时间累积错误和不确定反馈进行诊断。",
        "problem": "预训练视频模型并不会自动具备指令跟随、可控性、长时鲁棒性和部署能力；视频 rollout 成本高、错误随时间累积，异构反馈还可能不确定或区分度不足。",
        "contributions": [
          "把 supervised policy shaping、reward-driven RL 和 preference refinement 组织为统一的分阶段优化栈。",
          "将诊断机制和稳定性约束提升为视频模型后训练的一等组件。",
          "给出面向部署的流程，用于同时改善感知质量、时间一致性和 prompt adherence，并尽量保留初始化阶段已有的可控性。"
        ],
        "evidence": "TeleBoost 是系统化框架与报告，而不是单一算法。相关结论应理解为完整后训练 recipe 的综合效果，不能据此推断任一阶段单独产生全部属性。",
        "limitations": "该框架需要高成本视频 rollout、多种反馈源、诊断基础设施和分阶段调参。复现与比较必须说明完整训练栈和初始化条件。",
        "positioning": "TeleBoost 适合用于讨论面向生产的视频生成对齐与后训练系统。与只提出某个优化器的工作不同，它强调 SFT、RL 和偏好阶段之间的编排与稳定性。",
        "citation_ready": "Liang 等提出 TeleBoost，将 supervised policy shaping、reward-driven reinforcement learning 与 preference refinement 组织为受稳定性约束的视频后训练框架，以提升画面质量、可控性、时间一致性和指令遵循。"
      }
    },
    "citation": {
      "id": "liang2026teleboost",
      "type": "article",
      "title": "TeleBoost: A Systematic Alignment Framework for High-Fidelity, Controllable, and Robust Video Generation",
      "author": [
        {
          "given": "Yuanzhi",
          "family": "Liang"
        },
        {
          "given": "Xuan'er",
          "family": "Wu"
        },
        {
          "given": "Yirui",
          "family": "Liu"
        },
        {
          "given": "Yijie",
          "family": "Fang"
        },
        {
          "given": "Yizhen",
          "family": "Fan"
        },
        {
          "given": "Ke",
          "family": "Hao"
        },
        {
          "given": "Rui",
          "family": "Li"
        },
        {
          "given": "Ruiying",
          "family": "Liu"
        },
        {
          "given": "Ziqi",
          "family": "Ni"
        },
        {
          "given": "Peng",
          "family": "Yu"
        },
        {
          "given": "Yanbo",
          "family": "Wang"
        },
        {
          "given": "Haibin",
          "family": "Huang"
        },
        {
          "given": "Qizhen",
          "family": "Weng"
        },
        {
          "given": "Chi",
          "family": "Zhang"
        },
        {
          "given": "Xuelong",
          "family": "Li"
        }
      ],
      "container-title": "arXiv",
      "issued": {
        "date-parts": [
          [
            2026,
            2,
            7
          ]
        ]
      },
      "URL": "https://arxiv.org/abs/2602.07595",
      "abstract": "Post-training is the decisive step for converting a pretrained video generator into a production-oriented model that is instruction-following, controllable, and robust over long temporal horizons. This report presents a systematical post-training framework that organizes supervised policy shaping, reward-driven reinforcement learning, and preference-based refinement into a single stability-constrained optimization stack. The framework is designed around practical video-generation constraints, including high rollout cost, temporally compounding failure modes, and feedback that is heterogeneous, uncertain, and often weakly discriminative. By treating optimization as a staged, diagnostic-driven process rather than a collection of isolated tricks, the report summarizes a cohesive recipe for improving perceptual fidelity, temporal coherence, and prompt adherence while preserving the controllability established at initialization. The resulting framework provides a clear blueprint for building scalable post-training pipelines that remain stable, extensible, and effective in real-world deployment settings.",
      "keyword": "video generation, post-training, alignment, reinforcement learning, preference optimization",
      "archive": "arXiv",
      "archive_location": "2602.07595",
      "genre": "Preprint"
    },
    "resources": [
      {
        "label": "Project",
        "url": "https://tele-ai.github.io/TeleBoost/"
      },
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2602.07595"
      },
      {
        "label": "PDF",
        "url": "https://arxiv.org/pdf/2602.07595"
      }
    ]
  },
  {
    "schema_version": 1,
    "slug": "rl-vgm",
    "canonical_url": "https://akira-l.github.io/publications/rl-vgm/",
    "language_urls": {
      "en": "https://akira-l.github.io/publications/rl-vgm/",
      "zh-CN": "https://akira-l.github.io/zh/publications/rl-vgm/"
    },
    "title": "Integrating reinforcement learning with visual generative models: foundations and advances",
    "short_title": "RL for Visual Generative Models",
    "authors": [
      "Yuanzhi Liang",
      "Yijie Fang",
      "Rui Li",
      "Ziqi Ni",
      "Ruijie Su",
      "Chi Zhang"
    ],
    "publication": {
      "kind": "journal",
      "venue": "Vicinagearth",
      "citation_container_title": "Vicinagearth",
      "status": "published",
      "year": 2026,
      "publication_date": "2026-01-29",
      "publisher": "Springer Nature",
      "volume": "3",
      "issue": "1",
      "article_number": "2"
    },
    "identifiers": {
      "doi": "10.1007/s44336-025-00030-z",
      "arxiv": "2508.10316",
      "arxiv_primary_class": "cs.CV"
    },
    "arxiv_dates": {
      "first_posted": "2025-08-14",
      "last_revised": "2026-01-19"
    },
    "keywords": [
      "reinforcement learning",
      "visual generative models",
      "image generation",
      "video generation",
      "3D and 4D generation",
      "survey"
    ],
    "official_abstract": "Generative models have made significant progress in synthesizing visual content, including images, videos, and 3D/4D structures. However, they are typically trained with surrogate objectives such as likelihood or reconstruction loss, which often misalign with perceptual quality, semantic accuracy, or physical realism. Reinforcement learning (RL) offers a principled framework for optimizing non-differentiable, preference-driven, and temporally structured objectives. Recent advances demonstrate its effectiveness in enhancing controllability, consistency, and human alignment across generative tasks. This survey provides a systematic overview of RL-based methods for visual content generation. We review the evolution of RL from classical control to its role as a general-purpose optimization tool, and examine its integration into image, video, and 3D/4D generation. Across these domains, RL serves not only as a fine-tuning mechanism but also as a structural component for aligning generation with complex, high-level goals. We conclude with open challenges and future research directions at the intersection of RL and generative modeling.",
    "official_abstract_rights": "Excluded from the site's CC BY 4.0 license; original paper rights apply.",
    "primary_source": {
      "label": "Springer journal article",
      "url": "https://link.springer.com/article/10.1007/s44336-025-00030-z",
      "version": "Version of record, published 2026-01-29",
      "method_locator": "Abstract; sections on RL evolution and image, video, and 3D/4D generation",
      "evidence_locator": "Abstract; domain survey tables and discussion sections"
    },
    "source_checked": "2026-07-31",
    "verification_status": "author-verified",
    "bibliographic_note": "This citation describes the six-author Vicinagearth version of record. arXiv:2508.10316v3 lists Ke Hao as an additional third author, so the arXiv identifier is retained only as a related version and is not mixed into the journal citation exports.",
    "author_verified_on": "2026-07-31",
    "commentary": {
      "license": "https://creativecommons.org/licenses/by/4.0/",
      "en": {
        "summary": "This survey organizes reinforcement learning for image, video, and 3D/4D generation, treating RL not only as a fine-tuning algorithm but as an interface for optimizing non-differentiable, preference-driven, temporal, and high-level objectives.",
        "problem": "Likelihood and reconstruction objectives are useful training surrogates but can diverge from perceptual quality, semantic accuracy, physical realism, controllability, and human preferences across visual generation tasks.",
        "contributions": [
          "Traces the evolution of RL from classical control toward a general optimization and alignment framework.",
          "Systematizes RL integrations across image, video, and 3D/4D generation.",
          "Identifies cross-domain challenges and future directions at the intersection of RL and visual generative modeling."
        ],
        "evidence": "As a survey, the paper synthesizes and categorizes prior literature rather than claiming a new model's benchmark improvement. Its tables, taxonomy, domain sections, and discussion of open problems are the relevant evidence.",
        "limitations": "The area changes rapidly, so coverage is bounded by the review's search period and inclusion criteria. Readers should use the version of record and its bibliography to verify whether later methods alter the taxonomy or conclusions.",
        "positioning": "This work can serve as a high-level citation for the overall role of RL in visual generation and as a navigation source for domain-specific literature in images, videos, and 3D/4D content.",
        "citation_ready": "Liang et al. survey reinforcement learning for visual generative models, organizing methods across image, video, and 3D/4D generation and framing RL as a general mechanism for optimizing preference-driven, non-differentiable, and structured objectives."
      },
      "zh-CN": {
        "summary": "这篇综述系统整理图像、视频与 3D/4D 生成中的 reinforcement learning，并把 RL 视为连接不可微目标、偏好反馈、时间结构和高层目标的通用优化接口，而不仅是某种 fine-tuning 算法。",
        "problem": "Likelihood 和 reconstruction loss 是常用代理目标，但它们可能与视觉质量、语义准确性、物理真实性、可控性和人类偏好不一致。",
        "contributions": [
          "梳理 RL 从经典控制到通用优化与对齐框架的演化。",
          "系统组织 RL 在图像、视频和 3D/4D 生成中的集成方式。",
          "总结跨领域共同挑战以及 RL 与视觉生成交叉方向的未来问题。"
        ],
        "evidence": "作为综述，本文的证据来自对既有文献的分类、归纳和比较，而不是某个新模型的单一 benchmark 提升；应重点参考其 taxonomy、领域章节、表格和开放问题讨论。",
        "limitations": "该方向变化很快，覆盖范围受检索时间和纳入标准约束。后续使用时应以正式版本及其参考文献为入口，检查新工作是否改变已有分类或结论。",
        "positioning": "这项工作可作为“RL 在视觉生成中的总体角色”的高层引用，也可作为进入图像、视频和 3D/4D 各子方向文献的导航来源。",
        "citation_ready": "Liang 等系统综述了视觉生成模型中的 reinforcement learning，覆盖图像、视频和 3D/4D 生成，并将 RL 概括为优化偏好驱动、不可微和结构化目标的通用机制。"
      }
    },
    "citation": {
      "id": "liang2026rlvisualgeneration",
      "type": "article-journal",
      "title": "Integrating reinforcement learning with visual generative models: foundations and advances",
      "author": [
        {
          "given": "Yuanzhi",
          "family": "Liang"
        },
        {
          "given": "Yijie",
          "family": "Fang"
        },
        {
          "given": "Rui",
          "family": "Li"
        },
        {
          "given": "Ziqi",
          "family": "Ni"
        },
        {
          "given": "Ruijie",
          "family": "Su"
        },
        {
          "given": "Chi",
          "family": "Zhang"
        }
      ],
      "container-title": "Vicinagearth",
      "issued": {
        "date-parts": [
          [
            2026,
            1,
            29
          ]
        ]
      },
      "URL": "https://doi.org/10.1007/s44336-025-00030-z",
      "abstract": "Generative models have made significant progress in synthesizing visual content, including images, videos, and 3D/4D structures. However, they are typically trained with surrogate objectives such as likelihood or reconstruction loss, which often misalign with perceptual quality, semantic accuracy, or physical realism. Reinforcement learning (RL) offers a principled framework for optimizing non-differentiable, preference-driven, and temporally structured objectives. Recent advances demonstrate its effectiveness in enhancing controllability, consistency, and human alignment across generative tasks. This survey provides a systematic overview of RL-based methods for visual content generation. We review the evolution of RL from classical control to its role as a general-purpose optimization tool, and examine its integration into image, video, and 3D/4D generation. Across these domains, RL serves not only as a fine-tuning mechanism but also as a structural component for aligning generation with complex, high-level goals. We conclude with open challenges and future research directions at the intersection of RL and generative modeling.",
      "keyword": "reinforcement learning, visual generative models, image generation, video generation, 3D and 4D generation, survey",
      "publisher": "Springer Nature",
      "DOI": "10.1007/s44336-025-00030-z",
      "volume": "3",
      "issue": "1",
      "page": "2"
    },
    "resources": [
      {
        "label": "Paper",
        "url": "https://link.springer.com/article/10.1007/s44336-025-00030-z"
      },
      {
        "label": "Project",
        "url": "https://visgenrlsurvey.liangyzh18.workers.dev/"
      },
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2508.10316"
      }
    ]
  },
  {
    "schema_version": 1,
    "slug": "vipo",
    "canonical_url": "https://akira-l.github.io/publications/vipo/",
    "language_urls": {
      "en": "https://akira-l.github.io/publications/vipo/",
      "zh-CN": "https://akira-l.github.io/zh/publications/vipo/"
    },
    "title": "Seeing What Matters: Visual Preference Policy Optimization for Visual Generation",
    "short_title": "ViPO",
    "authors": [
      "Ziqi Ni",
      "Yuanzhi Liang",
      "Rui Li",
      "Yi Zhou",
      "Haibin Huang",
      "Chi Zhang",
      "Xuelong Li"
    ],
    "publication": {
      "kind": "conference",
      "venue": "IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR 2026)",
      "citation_container_title": "Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition",
      "status": "published",
      "year": 2026,
      "publication_date": "2026",
      "pages": "27260-27269"
    },
    "identifiers": {
      "arxiv": "2511.18719",
      "arxiv_primary_class": "cs.CV"
    },
    "arxiv_dates": {
      "first_posted": "2025-11-24",
      "last_revised": "2026-05-15"
    },
    "keywords": [
      "visual generation",
      "GRPO",
      "pixel-level advantage",
      "preference optimization",
      "structured feedback"
    ],
    "official_abstract": "Reinforcement learning (RL) has become a powerful tool for post-training visual generative models, with Group Relative Policy Optimization (GRPO) increasingly used to align generators with human preferences. However, existing GRPO pipelines rely on a single scalar reward per sample, treating each image or video as a holistic entity and ignoring the rich spatial and temporal structure of visual content. This coarse supervision hinders the correction of localized artifacts and the modeling of fine-grained perceptual cues. We introduce Visual Preference Policy Optimization (ViPO), a GRPO variant that lifts scalar feedback into structured, pixel-level advantages. ViPO employs a Perceptual Structuring Module that uses pretrained vision backbones to construct spatially and temporally aware advantage maps, redistributing optimization pressure toward perceptually important regions while preserving the stability of standard GRPO. Across both image and video benchmarks, ViPO consistently outperforms vanilla GRPO, improving in-domain alignment with human-preference rewards and enhancing generalization on out-of-domain evaluations. The method is architecture-agnostic, lightweight, and fully compatible with existing GRPO training pipelines, providing a more expressive and informative learning signal for visual generation.",
    "official_abstract_rights": "Excluded from the site's CC BY 4.0 license; original paper rights apply.",
    "primary_source": {
      "label": "CVF open-access paper",
      "url": "https://openaccess.thecvf.com/content/CVPR2026/html/Ni_Seeing_What_Matters_Visual_Preference_Policy_Optimization_for_Visual_Generation_CVPR_2026_paper.html",
      "version": "CVPR 2026 open-access version; arXiv:2511.18719v4 checked for revision date",
      "method_locator": "Abstract; Perceptual Structuring Module section",
      "evidence_locator": "Abstract; image and video benchmark sections"
    },
    "source_checked": "2026-07-31",
    "verification_status": "author-verified",
    "author_verified_on": "2026-07-31",
    "commentary": {
      "license": "https://creativecommons.org/licenses/by/4.0/",
      "en": {
        "summary": "ViPO turns one scalar reward per generated sample into spatially and temporally structured, pixel-level advantage maps, using pretrained vision features to focus GRPO updates on perceptually important regions while retaining the standard training pipeline.",
        "problem": "A scalar reward treats an image or video as one indivisible unit, so GRPO cannot explicitly direct optimization toward localized artifacts or fine-grained spatial and temporal cues.",
        "contributions": [
          "Introduces a GRPO variant that lifts scalar feedback into pixel-level structured advantages.",
          "Uses a Perceptual Structuring Module with pretrained vision backbones to build spatially and temporally aware advantage maps.",
          "Keeps the method lightweight, architecture-agnostic, and compatible with existing GRPO pipelines."
        ],
        "evidence": "The paper reports improvements over vanilla GRPO on both image and video benchmarks, including in-domain preference alignment and out-of-domain generalization. Consult the official tables for exact rewards, datasets, backbones, and effect sizes.",
        "limitations": "The structured map is induced by pretrained visual features and therefore depends on what those backbones encode. Pixel-level optimization does not by itself guarantee that every reward model is reliable or causally localized.",
        "positioning": "ViPO belongs to work that increases reward granularity for generative-model RL. Its defining step is spatial–temporal redistribution of advantage rather than changing the base generator or replacing GRPO's overall optimization structure.",
        "citation_ready": "Ni et al. propose Visual Preference Policy Optimization (ViPO), which uses a perceptual structuring module to transform scalar rewards into spatially and temporally aware pixel-level advantages for GRPO-based image and video generation."
      },
      "zh-CN": {
        "summary": "ViPO 把每个生成样本的单一标量 reward 转换为空间和时间结构化的像素级 advantage map，利用预训练视觉特征把 GRPO 更新集中到感知上重要的区域，同时保留标准训练流程。",
        "problem": "标量奖励把整张图像或整段视频视为不可分割的整体，GRPO 因而难以针对局部伪影或细粒度时空线索进行定向优化。",
        "contributions": [
          "提出把标量反馈提升为像素级结构化 advantage 的 GRPO 变体。",
          "利用预训练视觉 backbone 构建具有空间和时间感知能力的 advantage map。",
          "保持 architecture-agnostic、轻量并兼容现有 GRPO pipeline。"
        ],
        "evidence": "论文报告在图像和视频 benchmark 上优于 vanilla GRPO，同时改善域内偏好对齐与域外泛化。准确奖励、数据集、backbone 和提升幅度应以正式论文表格为准。",
        "limitations": "结构化 advantage map 由预训练视觉特征诱导，因此受这些 backbone 表征能力的影响；像素级优化也不能自动保证每个 reward model 都可靠或具备真实的局部因果性。",
        "positioning": "ViPO 属于提高生成模型 RL 反馈粒度的工作。其关键不是替换生成器或改变 GRPO 总体框架，而是对 advantage 进行时空重分配。",
        "citation_ready": "Ni 等提出 Visual Preference Policy Optimization（ViPO），通过 perceptual structuring module 将标量奖励转换为具备时空感知的像素级优势，用于基于 GRPO 的图像和视频生成。"
      }
    },
    "citation": {
      "id": "ni2026vipo",
      "type": "paper-conference",
      "title": "Seeing What Matters: Visual Preference Policy Optimization for Visual Generation",
      "author": [
        {
          "given": "Ziqi",
          "family": "Ni"
        },
        {
          "given": "Yuanzhi",
          "family": "Liang"
        },
        {
          "given": "Rui",
          "family": "Li"
        },
        {
          "given": "Yi",
          "family": "Zhou"
        },
        {
          "given": "Haibin",
          "family": "Huang"
        },
        {
          "given": "Chi",
          "family": "Zhang"
        },
        {
          "given": "Xuelong",
          "family": "Li"
        }
      ],
      "container-title": "Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition",
      "issued": {
        "date-parts": [
          [
            2026
          ]
        ]
      },
      "URL": "https://openaccess.thecvf.com/content/CVPR2026/html/Ni_Seeing_What_Matters_Visual_Preference_Policy_Optimization_for_Visual_Generation_CVPR_2026_paper.html",
      "abstract": "Reinforcement learning (RL) has become a powerful tool for post-training visual generative models, with Group Relative Policy Optimization (GRPO) increasingly used to align generators with human preferences. However, existing GRPO pipelines rely on a single scalar reward per sample, treating each image or video as a holistic entity and ignoring the rich spatial and temporal structure of visual content. This coarse supervision hinders the correction of localized artifacts and the modeling of fine-grained perceptual cues. We introduce Visual Preference Policy Optimization (ViPO), a GRPO variant that lifts scalar feedback into structured, pixel-level advantages. ViPO employs a Perceptual Structuring Module that uses pretrained vision backbones to construct spatially and temporally aware advantage maps, redistributing optimization pressure toward perceptually important regions while preserving the stability of standard GRPO. Across both image and video benchmarks, ViPO consistently outperforms vanilla GRPO, improving in-domain alignment with human-preference rewards and enhancing generalization on out-of-domain evaluations. The method is architecture-agnostic, lightweight, and fully compatible with existing GRPO training pipelines, providing a more expressive and informative learning signal for visual generation.",
      "keyword": "visual generation, GRPO, pixel-level advantage, preference optimization, structured feedback",
      "page": "27260-27269"
    },
    "resources": [
      {
        "label": "Paper",
        "url": "https://openaccess.thecvf.com/content/CVPR2026/html/Ni_Seeing_What_Matters_Visual_Preference_Policy_Optimization_for_Visual_Generation_CVPR_2026_paper.html"
      },
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2511.18719"
      },
      {
        "label": "PDF",
        "url": "https://arxiv.org/pdf/2511.18719"
      }
    ]
  },
  {
    "schema_version": 1,
    "slug": "rats",
    "canonical_url": "https://akira-l.github.io/publications/rats/",
    "language_urls": {
      "en": "https://akira-l.github.io/publications/rats/",
      "zh-CN": "https://akira-l.github.io/zh/publications/rats/"
    },
    "title": "Reward-Aware Trajectory Shaping for Few-step Visual Generation",
    "short_title": "RATS",
    "authors": [
      "Rui Li",
      "Bingyu Li",
      "Yuanzhi Liang",
      "Haibin Huang",
      "Chi Zhang",
      "XueLong Li"
    ],
    "publication": {
      "kind": "conference",
      "venue": "ACM Multimedia 2026 (accepted)",
      "citation_container_title": "34th ACM International Conference on Multimedia (ACM Multimedia 2026)",
      "status": "forthcoming",
      "year": 2026,
      "publication_date": "2026",
      "publisher": "ACM"
    },
    "identifiers": {
      "arxiv": "2604.14910",
      "arxiv_primary_class": "cs.CV"
    },
    "arxiv_dates": {
      "first_posted": "2026-04-16",
      "last_revised": "2026-04-27"
    },
    "keywords": [
      "few-step generation",
      "trajectory distillation",
      "preference alignment",
      "reward-aware gating",
      "diffusion models"
    ],
    "official_abstract": "Achieving high-fidelity generation in extremely few sampling steps has long been a central goal of generative modeling. Existing approaches largely rely on distillation-based frameworks to compress the original multi-step denoising process into a few-step generator. However, such methods inherently constrain the student to imitate a stronger multi-step teacher, imposing the teacher as an upper bound on student performance. We argue that introducing preference alignment awareness enables the student to optimize toward reward-preferred generation quality, potentially surpassing the teacher instead of being restricted to rigid teacher imitation. To this end, we propose Reward-Aware Trajectory Shaping (RATS), a lightweight framework for preference-aligned few-step generation. Specifically, teacher and student latent trajectories are aligned at key denoising stages through horizon matching, while a reward-aware gate is introduced to adaptively regulate teacher guidance based on their relative reward performance. Trajectory shaping is strengthened when the teacher achieves higher rewards, and relaxed when the student matches or surpasses the teacher, thereby enabling continued reward-driven improvement. By seamlessly integrating trajectory distillation, reward-aware gating, and preference alignment, RATS effectively transfers preference-relevant knowledge from high-step generators without incurring additional test-time computational overhead. Experimental results demonstrate that RATS substantially improves the efficiency--quality trade-off in few-step visual generation, significantly narrowing the gap between few-step students and stronger multi-step generators.",
    "official_abstract_rights": "Excluded from the site's CC BY 4.0 license; original paper rights apply.",
    "primary_source": {
      "label": "arXiv record",
      "url": "https://arxiv.org/abs/2604.14910",
      "version": "arXiv:2604.14910v3",
      "method_locator": "Abstract; method sections on horizon matching and the reward-aware gate",
      "evidence_locator": "Abstract; few-step generation experiments"
    },
    "source_checked": "2026-07-31",
    "verification_status": "author-verified",
    "bibliographic_note": "The status “accepted at ACM Multimedia 2026” is author-supplied. The paper-level proceedings record, DOI, pagination, and final ACM citation were not yet public on 2026-07-31; this export is explicitly marked forthcoming and includes the current arXiv identifier.",
    "author_verified_on": "2026-07-31",
    "commentary": {
      "license": "https://creativecommons.org/licenses/by/4.0/",
      "en": {
        "summary": "RATS combines trajectory distillation with preference feedback: horizon matching aligns teacher and student at key denoising stages, while a reward-aware gate strengthens teacher guidance only when the teacher is better under the chosen reward.",
        "problem": "Rigid distillation makes a few-step student imitate a multi-step teacher and can turn the teacher into a performance ceiling, even when preference optimization indicates that the student could improve beyond it.",
        "contributions": [
          "Aligns teacher and student latent trajectories at key stages through horizon matching.",
          "Uses a reward-aware gate to strengthen or relax teacher guidance according to relative reward performance.",
          "Combines distillation and preference alignment without adding test-time computation."
        ],
        "evidence": "The paper reports an improved efficiency–quality trade-off and a narrower gap to stronger multi-step generators. Exact step counts, base models, rewards, and metric values should be taken from the current paper tables rather than inferred from this summary.",
        "limitations": "The gate inherits the assumptions and possible biases of the reward signal used to compare teacher and student. The paper targets few-step visual generation; transfer to other distillation settings requires separate validation.",
        "positioning": "RATS sits between trajectory distillation and reward-based alignment. Its main distinction is that teacher imitation is conditional on relative reward quality instead of being enforced as a fixed target.",
        "citation_ready": "Li et al. introduce Reward-Aware Trajectory Shaping (RATS), combining horizon-matched teacher–student trajectories with a reward-aware gate that conditionally adjusts teacher guidance for preference-aligned few-step visual generation."
      },
      "zh-CN": {
        "summary": "RATS 把轨迹蒸馏与偏好反馈结合起来：horizon matching 在关键去噪阶段对齐 teacher 与 student，reward-aware gate 只在 teacher 的奖励表现更好时加强指导。",
        "problem": "固定蒸馏要求 few-step student 持续模仿 multi-step teacher，容易把 teacher 变成性能上限，即使偏好奖励表明 student 已经能够达到或超过 teacher。",
        "contributions": [
          "通过 horizon matching 对齐 teacher 和 student 在关键阶段的 latent trajectory。",
          "根据双方相对 reward 表现动态增强或减弱 teacher guidance。",
          "在不增加推理开销的前提下结合轨迹蒸馏与 preference alignment。"
        ],
        "evidence": "论文报告 few-step 生成在效率—质量权衡上获得改进，并缩小与强 multi-step generator 的差距。具体步数、底模、奖励和指标数值应查阅当前版本的实验表格。",
        "limitations": "Reward-aware gate 会继承用于比较 teacher 与 student 的奖励信号所包含的假设与偏差；方法面向 few-step visual generation，迁移到其他蒸馏设置仍需验证。",
        "positioning": "RATS 位于 trajectory distillation 与 reward-based alignment 的交叉点。它的关键差异是根据相对奖励质量决定是否跟随 teacher，而不是始终把 teacher 当作固定目标。",
        "citation_ready": "Li 等提出 Reward-Aware Trajectory Shaping（RATS），利用 horizon matching 对齐师生轨迹，并通过 reward-aware gate 有条件地调节 teacher guidance，以实现偏好对齐的少步视觉生成。"
      }
    },
    "citation": {
      "id": "li2026rats",
      "type": "paper-conference",
      "title": "Reward-Aware Trajectory Shaping for Few-step Visual Generation",
      "author": [
        {
          "given": "Rui",
          "family": "Li"
        },
        {
          "given": "Bingyu",
          "family": "Li"
        },
        {
          "given": "Yuanzhi",
          "family": "Liang"
        },
        {
          "given": "Haibin",
          "family": "Huang"
        },
        {
          "given": "Chi",
          "family": "Zhang"
        },
        {
          "given": "XueLong",
          "family": "Li"
        }
      ],
      "container-title": "34th ACM International Conference on Multimedia (ACM Multimedia 2026)",
      "issued": {
        "date-parts": [
          [
            2026
          ]
        ]
      },
      "URL": "https://arxiv.org/abs/2604.14910",
      "abstract": "Achieving high-fidelity generation in extremely few sampling steps has long been a central goal of generative modeling. Existing approaches largely rely on distillation-based frameworks to compress the original multi-step denoising process into a few-step generator. However, such methods inherently constrain the student to imitate a stronger multi-step teacher, imposing the teacher as an upper bound on student performance. We argue that introducing preference alignment awareness enables the student to optimize toward reward-preferred generation quality, potentially surpassing the teacher instead of being restricted to rigid teacher imitation. To this end, we propose Reward-Aware Trajectory Shaping (RATS), a lightweight framework for preference-aligned few-step generation. Specifically, teacher and student latent trajectories are aligned at key denoising stages through horizon matching, while a reward-aware gate is introduced to adaptively regulate teacher guidance based on their relative reward performance. Trajectory shaping is strengthened when the teacher achieves higher rewards, and relaxed when the student matches or surpasses the teacher, thereby enabling continued reward-driven improvement. By seamlessly integrating trajectory distillation, reward-aware gating, and preference alignment, RATS effectively transfers preference-relevant knowledge from high-step generators without incurring additional test-time computational overhead. Experimental results demonstrate that RATS substantially improves the efficiency--quality trade-off in few-step visual generation, significantly narrowing the gap between few-step students and stronger multi-step generators.",
      "keyword": "few-step generation, trajectory distillation, preference alignment, reward-aware gating, diffusion models",
      "publisher": "ACM",
      "archive": "arXiv",
      "archive_location": "2604.14910",
      "genre": "Forthcoming conference paper",
      "status": "forthcoming"
    },
    "resources": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2604.14910"
      },
      {
        "label": "PDF",
        "url": "https://arxiv.org/pdf/2604.14910"
      }
    ]
  },
  {
    "schema_version": 1,
    "slug": "taros",
    "canonical_url": "https://akira-l.github.io/publications/taros/",
    "language_urls": {
      "en": "https://akira-l.github.io/publications/taros/",
      "zh-CN": "https://akira-l.github.io/zh/publications/taros/"
    },
    "title": "Rethinking Reward Signals in Video GRPO: When Scores Become Targets",
    "short_title": "TaRoS",
    "authors": [
      "Rui Li",
      "Yuanzhi Liang",
      "Ziqi Ni",
      "Haibin Huang",
      "Chi Zhang",
      "Xuelong Li"
    ],
    "publication": {
      "kind": "conference",
      "venue": "European Conference on Computer Vision (ECCV 2026, accepted)",
      "citation_container_title": "European Conference on Computer Vision (ECCV 2026)",
      "status": "forthcoming",
      "year": 2026,
      "publication_date": "2026"
    },
    "identifiers": {
      "arxiv": "2511.19356",
      "arxiv_primary_class": "cs.CV"
    },
    "arxiv_dates": {
      "first_posted": "2025-11-24",
      "last_revised": "2026-07-17"
    },
    "keywords": [
      "video generation",
      "GRPO",
      "reward saturation",
      "reward hacking",
      "Goodhart's law"
    ],
    "official_abstract": "Group Relative Policy Optimization (GRPO) enables stable and preference-oriented updates via group-wise comparisons for post-training video generation. However, GRPO directly optimizes reward-induced advantages. Under sustained optimization, the reward score can lose fidelity as a proxy for true video quality, consistent with the phenomenon described by Goodhart's Law. This leads to two recurring issues: (i) shortcut-driven optimization under composite objectives and (ii) reward saturation within prompt groups. To address these issues, we introduce TaRoS, a Target-Robust Reward Signaling framework for Video generation GRPO. TaRoS leverages component level performance assessment together with intra-group sparsity to organize multi-aspect rewards towards optimization objectives. In addition, it adaptively downweights components that exhibit saturation, thereby preserving effective optimization directions and mitigating redundancy. This maintains meaningful optimization directions and preserves within-group ranking separation, thereby preventing reward hacking and leading to more reliable policy updates. Extensive experiments show consistent improvements in visual fidelity, motion coherence, and text-video alignment over strong baselines.",
    "official_abstract_rights": "Excluded from the site's CC BY 4.0 license; original paper rights apply.",
    "primary_source": {
      "label": "arXiv record",
      "url": "https://arxiv.org/abs/2511.19356",
      "version": "arXiv:2511.19356v4, revised 2026-07-17",
      "method_locator": "Abstract; TaRoS sections on component-level assessment, intra-group sparsity, and saturation downweighting",
      "evidence_locator": "Abstract; video generation experiments in v4"
    },
    "source_checked": "2026-07-31",
    "verification_status": "author-verified",
    "bibliographic_note": "The status “accepted at ECCV 2026” is author-supplied. The paper-level Springer/ECVA proceedings record, DOI, volume, and pagination were not yet public on 2026-07-31; this export is explicitly marked forthcoming and includes the current arXiv identifier.",
    "author_verified_on": "2026-07-31",
    "commentary": {
      "license": "https://creativecommons.org/licenses/by/4.0/",
      "en": {
        "summary": "TaRoS v4 addresses reward hacking and saturation in video GRPO by organizing multi-aspect feedback with component-level performance assessment and intra-group sparsity, then adaptively downweighting components whose scores have saturated.",
        "problem": "When GRPO repeatedly optimizes reward-induced advantages, reward scores can stop being faithful proxies for true video quality: composite objectives invite shortcuts and scores within a prompt group may saturate, erasing useful rankings.",
        "contributions": [
          "Assesses performance at the reward-component level rather than treating a composite score as one stable target.",
          "Uses intra-group sparsity to organize multi-aspect rewards toward optimization objectives.",
          "Adaptively downweights saturated components to preserve effective directions and within-group ranking separation."
        ],
        "evidence": "The current arXiv v4 reports consistent improvements in visual fidelity, motion coherence, and text–video alignment over strong baselines. Earlier descriptions centered on generic target calibration should not be used for this version.",
        "limitations": "TaRoS diagnoses behavior through the available reward components, so missing or systematically biased aspects remain outside its correction mechanism. The public record is a revised preprint associated with an accepted ECCV 2026 paper; final proceedings metadata may change.",
        "positioning": "TaRoS belongs to robust reward signaling for video-generation RL. Unlike reward aggregation alone, it explicitly responds to Goodhart-style shortcut optimization and component saturation during continued GRPO training.",
        "citation_ready": "Li et al. propose TaRoS, a target-robust reward-signaling framework for video GRPO that combines component-level assessment, intra-group sparsity, and adaptive downweighting of saturated reward components to reduce reward hacking."
      },
      "zh-CN": {
        "summary": "TaRoS v4 针对 video GRPO 中的 reward hacking 与饱和：用 component-level performance assessment 和 intra-group sparsity 组织多维反馈，并自适应降低已经饱和的 reward component 权重。",
        "problem": "GRPO 持续优化 reward-induced advantage 后，reward score 可能不再忠实代表真实视频质量：复合目标会诱发 shortcut，prompt group 内的分数也可能饱和并失去排序信息。",
        "contributions": [
          "在 reward component 层面评估表现，而不是把复合分数视为始终稳定的目标。",
          "利用 intra-group sparsity 将多维奖励组织到优化目标上。",
          "自适应降低饱和 component 的权重，保留有效优化方向和组内排序差异。"
        ],
        "evidence": "当前 arXiv v4 报告相对强 baseline 在视觉质量、运动一致性和文本—视频对齐上的一致改进。针对旧版本的笼统“目标校准”描述不应继续用于 v4。",
        "limitations": "TaRoS 只能通过已有 reward component 诊断问题，未被奖励覆盖或系统性偏置的维度仍无法自动修正；当前公开记录是与 ECCV 2026 accepted paper 关联的修订预印本，最终 proceedings 信息可能变化。",
        "positioning": "TaRoS 属于视频生成 RL 的 robust reward signaling 工作。它不只是聚合奖励，而是直接处理持续 GRPO 优化中的 Goodhart-style shortcut 与 component saturation。",
        "citation_ready": "Li 等提出 TaRoS，一种面向 video GRPO 的 target-robust reward-signaling 框架，通过 component-level assessment、intra-group sparsity 和对饱和奖励分量的自适应降权减少 reward hacking。"
      }
    },
    "citation": {
      "id": "li2026taros",
      "type": "paper-conference",
      "title": "Rethinking Reward Signals in Video GRPO: When Scores Become Targets",
      "author": [
        {
          "given": "Rui",
          "family": "Li"
        },
        {
          "given": "Yuanzhi",
          "family": "Liang"
        },
        {
          "given": "Ziqi",
          "family": "Ni"
        },
        {
          "given": "Haibin",
          "family": "Huang"
        },
        {
          "given": "Chi",
          "family": "Zhang"
        },
        {
          "given": "Xuelong",
          "family": "Li"
        }
      ],
      "container-title": "European Conference on Computer Vision (ECCV 2026)",
      "issued": {
        "date-parts": [
          [
            2026
          ]
        ]
      },
      "URL": "https://arxiv.org/abs/2511.19356",
      "abstract": "Group Relative Policy Optimization (GRPO) enables stable and preference-oriented updates via group-wise comparisons for post-training video generation. However, GRPO directly optimizes reward-induced advantages. Under sustained optimization, the reward score can lose fidelity as a proxy for true video quality, consistent with the phenomenon described by Goodhart's Law. This leads to two recurring issues: (i) shortcut-driven optimization under composite objectives and (ii) reward saturation within prompt groups. To address these issues, we introduce TaRoS, a Target-Robust Reward Signaling framework for Video generation GRPO. TaRoS leverages component level performance assessment together with intra-group sparsity to organize multi-aspect rewards towards optimization objectives. In addition, it adaptively downweights components that exhibit saturation, thereby preserving effective optimization directions and mitigating redundancy. This maintains meaningful optimization directions and preserves within-group ranking separation, thereby preventing reward hacking and leading to more reliable policy updates. Extensive experiments show consistent improvements in visual fidelity, motion coherence, and text-video alignment over strong baselines.",
      "keyword": "video generation, GRPO, reward saturation, reward hacking, Goodhart's law",
      "archive": "arXiv",
      "archive_location": "2511.19356",
      "genre": "Forthcoming conference paper",
      "status": "forthcoming"
    },
    "resources": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2511.19356"
      },
      {
        "label": "PDF",
        "url": "https://arxiv.org/pdf/2511.19356"
      }
    ]
  },
  {
    "schema_version": 1,
    "slug": "otca",
    "canonical_url": "https://akira-l.github.io/publications/otca/",
    "language_urls": {
      "en": "https://akira-l.github.io/publications/otca/",
      "zh-CN": "https://akira-l.github.io/zh/publications/otca/"
    },
    "title": "Learning to Credit the Right Steps: Objective-aware Process Optimization for Visual Generation",
    "short_title": "OTCA",
    "authors": [
      "Rui Li",
      "Ke Hao",
      "Yuanzhi Liang",
      "Haibin Huang",
      "Chi Zhang",
      "Yun Gu",
      "XueLong Li"
    ],
    "publication": {
      "kind": "conference",
      "venue": "ACM Multimedia 2026 (accepted)",
      "citation_container_title": "34th ACM International Conference on Multimedia (ACM Multimedia 2026)",
      "status": "forthcoming",
      "year": 2026,
      "publication_date": "2026",
      "publisher": "ACM"
    },
    "identifiers": {
      "arxiv": "2604.19234",
      "arxiv_primary_class": "cs.CV"
    },
    "arxiv_dates": {
      "first_posted": "2026-04-21",
      "last_revised": "2026-04-27"
    },
    "keywords": [
      "visual generation",
      "GRPO",
      "credit assignment",
      "multi-objective optimization",
      "diffusion models"
    ],
    "official_abstract": "Reinforcement learning, particularly Group Relative Policy Optimization (GRPO), has emerged as an effective framework for post-training visual generative models with human preference signals. However, its effectiveness is fundamentally limited by coarse reward credit assignment. In modern visual generation, multiple reward models are often used to capture heterogeneous objectives, such as visual quality, motion consistency, and text alignment. Existing GRPO pipelines typically collapse these rewards into a single static scalar and propagate it uniformly across the entire diffusion trajectory. This design ignores the stage-specific roles of different denoising steps and produces mistimed or incompatible optimization signals. To address this issue, we propose Objective-aware Trajectory Credit Assignment (OTCA), a structured framework for fine-grained GRPO training. OTCA consists of two key components. Trajectory-Level Credit Decomposition estimates the relative importance of different denoising steps. Multi-Objective Credit Allocation adaptively weights and combines multiple reward signals throughout the denoising process. By jointly modeling temporal credit and objective-level credit, OTCA converts coarse reward supervision into a structured, timestep-aware training signal that better matches the iterative nature of diffusion-based generation. Extensive experiments show that OTCA consistently improves both image and video generation quality across evaluation metrics.",
    "official_abstract_rights": "Excluded from the site's CC BY 4.0 license; original paper rights apply.",
    "primary_source": {
      "label": "arXiv record",
      "url": "https://arxiv.org/abs/2604.19234",
      "version": "arXiv:2604.19234v2",
      "method_locator": "Abstract; method sections on Trajectory-Level Credit Decomposition and Multi-Objective Credit Allocation",
      "evidence_locator": "Abstract; image and video experiments"
    },
    "source_checked": "2026-07-31",
    "verification_status": "author-verified",
    "bibliographic_note": "The status “accepted at ACM Multimedia 2026” is author-supplied. The paper-level proceedings record, DOI, pagination, and final ACM citation were not yet public on 2026-07-31; this export is explicitly marked forthcoming and includes the current arXiv identifier.",
    "author_verified_on": "2026-07-31",
    "commentary": {
      "license": "https://creativecommons.org/licenses/by/4.0/",
      "en": {
        "summary": "OTCA replaces uniform, scalar reward propagation in visual GRPO with structured credit assignment along two axes: which denoising steps matter and which reward objectives should matter at each point in the trajectory.",
        "problem": "Visual GRPO commonly collapses heterogeneous rewards into one scalar and applies it uniformly to every denoising step, even though different stages play different roles and different objectives may become relevant at different times.",
        "contributions": [
          "Trajectory-Level Credit Decomposition estimates the relative importance of denoising steps instead of assigning equal credit across the trajectory.",
          "Multi-Objective Credit Allocation adaptively weights and combines heterogeneous rewards during denoising.",
          "The joint formulation produces a timestep-aware, objective-aware signal aligned with iterative diffusion generation."
        ],
        "evidence": "The authors report consistent improvements for both image and video generation across evaluation metrics. This page intentionally does not restate numerical gains; consult the paper's experimental tables for model-, dataset-, and metric-specific comparisons.",
        "limitations": "The method is formulated for GRPO-style post-training of diffusion-based visual generators and still relies on the coverage and validity of the underlying reward models. Accepted-conference metadata should be updated when final proceedings metadata becomes available.",
        "positioning": "OTCA is best positioned as a fine-grained credit-assignment method for visual-generation RL. It differs from methods that only aggregate multiple rewards or only reweight samples by jointly structuring credit over denoising time and reward objectives.",
        "citation_ready": "Li et al. propose Objective-aware Trajectory Credit Assignment (OTCA), which decomposes credit across denoising steps and adaptively allocates multiple reward objectives to provide structured supervision for GRPO-based image and video generation."
      },
      "zh-CN": {
        "summary": "OTCA 不再把一个标量 reward 平均传给所有去噪步，而是同时回答两个问题：轨迹中的哪些 step 更重要，以及每个阶段应该强调哪些 reward objective。",
        "problem": "视觉 GRPO 往往把画质、运动一致性、文本对齐等异构奖励压成一个静态标量，并对整个 diffusion trajectory 均匀传播，忽略了不同去噪阶段的职责差异。",
        "contributions": [
          "Trajectory-Level Credit Decomposition 估计不同去噪 step 的相对重要性。",
          "Multi-Objective Credit Allocation 在去噪过程中自适应组合多个奖励目标。",
          "两者联合把粗粒度 reward 转化为同时具备 timestep awareness 与 objective awareness 的训练信号。"
        ],
        "evidence": "论文报告了在图像和视频生成任务及多项评测指标上的一致改进。本页不转述具体数值，模型、数据集与指标层面的比较应以论文实验表格为准。",
        "limitations": "方法面向 diffusion visual generator 的 GRPO 后训练，仍然依赖底层 reward model 的覆盖范围和可靠性；会议正式 proceedings 发布后还需更新最终书目信息。",
        "positioning": "OTCA 适合被定位为视觉生成 RL 的细粒度 credit assignment 方法。它不是只做多奖励聚合或样本重权重，而是同时建模时间维度与目标维度的 credit。",
        "citation_ready": "Li 等提出 Objective-aware Trajectory Credit Assignment（OTCA），通过分解不同去噪步的贡献并自适应分配多种奖励目标，为基于 GRPO 的图像和视频生成提供结构化训练信号。"
      }
    },
    "citation": {
      "id": "li2026otca",
      "type": "paper-conference",
      "title": "Learning to Credit the Right Steps: Objective-aware Process Optimization for Visual Generation",
      "author": [
        {
          "given": "Rui",
          "family": "Li"
        },
        {
          "given": "Ke",
          "family": "Hao"
        },
        {
          "given": "Yuanzhi",
          "family": "Liang"
        },
        {
          "given": "Haibin",
          "family": "Huang"
        },
        {
          "given": "Chi",
          "family": "Zhang"
        },
        {
          "given": "Yun",
          "family": "Gu"
        },
        {
          "given": "XueLong",
          "family": "Li"
        }
      ],
      "container-title": "34th ACM International Conference on Multimedia (ACM Multimedia 2026)",
      "issued": {
        "date-parts": [
          [
            2026
          ]
        ]
      },
      "URL": "https://arxiv.org/abs/2604.19234",
      "abstract": "Reinforcement learning, particularly Group Relative Policy Optimization (GRPO), has emerged as an effective framework for post-training visual generative models with human preference signals. However, its effectiveness is fundamentally limited by coarse reward credit assignment. In modern visual generation, multiple reward models are often used to capture heterogeneous objectives, such as visual quality, motion consistency, and text alignment. Existing GRPO pipelines typically collapse these rewards into a single static scalar and propagate it uniformly across the entire diffusion trajectory. This design ignores the stage-specific roles of different denoising steps and produces mistimed or incompatible optimization signals. To address this issue, we propose Objective-aware Trajectory Credit Assignment (OTCA), a structured framework for fine-grained GRPO training. OTCA consists of two key components. Trajectory-Level Credit Decomposition estimates the relative importance of different denoising steps. Multi-Objective Credit Allocation adaptively weights and combines multiple reward signals throughout the denoising process. By jointly modeling temporal credit and objective-level credit, OTCA converts coarse reward supervision into a structured, timestep-aware training signal that better matches the iterative nature of diffusion-based generation. Extensive experiments show that OTCA consistently improves both image and video generation quality across evaluation metrics.",
      "keyword": "visual generation, GRPO, credit assignment, multi-objective optimization, diffusion models",
      "publisher": "ACM",
      "archive": "arXiv",
      "archive_location": "2604.19234",
      "genre": "Forthcoming conference paper",
      "status": "forthcoming"
    },
    "resources": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2604.19234"
      },
      {
        "label": "PDF",
        "url": "https://arxiv.org/pdf/2604.19234"
      }
    ]
  },
  {
    "schema_version": 1,
    "slug": "bpgo",
    "canonical_url": "https://akira-l.github.io/publications/bpgo/",
    "language_urls": {
      "en": "https://akira-l.github.io/publications/bpgo/",
      "zh-CN": "https://akira-l.github.io/zh/publications/bpgo/"
    },
    "title": "Learning What to Trust: Bayesian Prior-Guided Optimization for Visual Generation",
    "short_title": "BPGO",
    "authors": [
      "Ruiying Liu",
      "Yuanzhi Liang",
      "Haibin Huang",
      "Tianshu Yu",
      "Chi Zhang"
    ],
    "publication": {
      "kind": "conference",
      "venue": "IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR 2026)",
      "citation_container_title": "Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition",
      "status": "published",
      "year": 2026,
      "publication_date": "2026",
      "pages": "34408-34417"
    },
    "identifiers": {
      "arxiv": "2511.18919",
      "arxiv_primary_class": "cs.CV"
    },
    "arxiv_dates": {
      "first_posted": "2025-11-24",
      "last_revised": "2025-11-24"
    },
    "keywords": [
      "visual generation",
      "GRPO",
      "reward uncertainty",
      "Bayesian prior",
      "semantic alignment"
    ],
    "official_abstract": "Group Relative Policy Optimization (GRPO) has emerged as an effective and lightweight framework for post-training visual generative models. However, its performance is fundamentally limited by the ambiguity of textual visual correspondence: a single prompt may validly describe diverse visual outputs, and a single image or video may support multiple equally correct interpretations. This many to many relationship leads reward models to generate uncertain and weakly discriminative signals, causing GRPO to underutilize reliable feedback and overfit noisy ones. We introduce Bayesian Prior-Guided Optimization (BPGO), a novel extension of GRPO that explicitly models reward uncertainty through a semantic prior anchor. BPGO adaptively modulates optimization trust at two levels: inter-group Bayesian trust allocation emphasizes updates from groups consistent with the prior while down-weighting ambiguous ones, and intra-group prior-anchored renormalization sharpens sample distinctions by expanding confident deviations and compressing uncertain scores. Across both image and video generation tasks, BPGO delivers consistently stronger semantic alignment, enhanced perceptual fidelity, and faster convergence than standard GRPO and recent variants.",
    "official_abstract_rights": "Excluded from the site's CC BY 4.0 license; original paper rights apply.",
    "primary_source": {
      "label": "CVF open-access paper",
      "url": "https://openaccess.thecvf.com/content/CVPR2026/html/Liu_Learning_What_to_Trust_Bayesian_Prior-Guided_Optimization_for_Visual_Generation_CVPR_2026_paper.html",
      "version": "CVPR 2026 open-access version; arXiv:2511.18919v1 checked for first-posted metadata",
      "method_locator": "Abstract; inter-group trust allocation and intra-group renormalization sections",
      "evidence_locator": "Abstract; image and video experiments"
    },
    "source_checked": "2026-07-31",
    "verification_status": "author-verified",
    "author_verified_on": "2026-07-31",
    "commentary": {
      "license": "https://creativecommons.org/licenses/by/4.0/",
      "en": {
        "summary": "BPGO treats visual-reward scores as uncertain rather than equally trustworthy: a semantic prior anchors inter-group trust allocation and intra-group renormalization so that GRPO emphasizes confident feedback and suppresses ambiguous signals.",
        "problem": "Text and visual content have a many-to-many relationship, so reward models may assign uncertain or weakly discriminative scores; standard GRPO can then underuse reliable feedback and overfit noisy comparisons.",
        "contributions": [
          "Introduces a semantic prior anchor to model uncertainty in visual-generation rewards.",
          "Allocates trust across groups according to consistency with the prior.",
          "Renormalizes scores within each group to expand confident deviations and compress uncertain ones."
        ],
        "evidence": "The paper reports stronger semantic alignment, perceptual fidelity, and convergence than standard GRPO and recent variants on image and video generation. Exact baselines, metrics, and numerical differences should be cited from the official experiments.",
        "limitations": "BPGO's behavior depends on the suitability of its semantic prior and the reward models it calibrates. A prior can reduce ambiguity without establishing that every retained signal matches human judgment in all domains.",
        "positioning": "BPGO is a reward-uncertainty and trust-allocation extension of GRPO. It is distinct from methods that only change reward granularity or temporal credit because it calibrates which group- and sample-level feedback should be trusted.",
        "citation_ready": "Liu et al. introduce Bayesian Prior-Guided Optimization (BPGO), using a semantic prior to allocate trust across GRPO groups and renormalize within-group rewards for more reliable image and video generation post-training."
      },
      "zh-CN": {
        "summary": "BPGO 不把所有视觉 reward score 视为同样可信，而是用 semantic prior 同时锚定组间 trust allocation 与组内 renormalization，使 GRPO 强调置信反馈并抑制歧义信号。",
        "problem": "文本和视觉内容之间是多对多关系，reward model 容易产生不确定或区分度弱的分数，标准 GRPO 可能因此低估可靠反馈并过拟合噪声比较。",
        "contributions": [
          "引入 semantic prior anchor，显式建模视觉生成 reward 的不确定性。",
          "根据各组与先验的一致性在 group 之间分配优化信任。",
          "在 group 内重新归一化分数，放大可信偏差并压缩不确定分数。"
        ],
        "evidence": "论文报告在图像和视频生成中，相比标准 GRPO 及近期变体取得更强的语义对齐、感知质量和收敛速度。准确 baseline、指标和数值差异应引用正式实验。",
        "limitations": "BPGO 的行为依赖 semantic prior 是否合适以及底层 reward model 的质量。先验能够减少歧义，但不能证明所有保留信号在所有领域都符合人类判断。",
        "positioning": "BPGO 是 GRPO 的 reward uncertainty 与 trust allocation 扩展。它区别于只提高 reward 粒度或只做时间 credit 的方法，重点校准哪些 group 和 sample 级反馈值得信任。",
        "citation_ready": "Liu 等提出 Bayesian Prior-Guided Optimization（BPGO），利用 semantic prior 在 GRPO group 之间分配信任并重整组内奖励，以提高图像和视频生成后训练信号的可靠性。"
      }
    },
    "citation": {
      "id": "liu2026bpgo",
      "type": "paper-conference",
      "title": "Learning What to Trust: Bayesian Prior-Guided Optimization for Visual Generation",
      "author": [
        {
          "given": "Ruiying",
          "family": "Liu"
        },
        {
          "given": "Yuanzhi",
          "family": "Liang"
        },
        {
          "given": "Haibin",
          "family": "Huang"
        },
        {
          "given": "Tianshu",
          "family": "Yu"
        },
        {
          "given": "Chi",
          "family": "Zhang"
        }
      ],
      "container-title": "Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition",
      "issued": {
        "date-parts": [
          [
            2026
          ]
        ]
      },
      "URL": "https://openaccess.thecvf.com/content/CVPR2026/html/Liu_Learning_What_to_Trust_Bayesian_Prior-Guided_Optimization_for_Visual_Generation_CVPR_2026_paper.html",
      "abstract": "Group Relative Policy Optimization (GRPO) has emerged as an effective and lightweight framework for post-training visual generative models. However, its performance is fundamentally limited by the ambiguity of textual visual correspondence: a single prompt may validly describe diverse visual outputs, and a single image or video may support multiple equally correct interpretations. This many to many relationship leads reward models to generate uncertain and weakly discriminative signals, causing GRPO to underutilize reliable feedback and overfit noisy ones. We introduce Bayesian Prior-Guided Optimization (BPGO), a novel extension of GRPO that explicitly models reward uncertainty through a semantic prior anchor. BPGO adaptively modulates optimization trust at two levels: inter-group Bayesian trust allocation emphasizes updates from groups consistent with the prior while down-weighting ambiguous ones, and intra-group prior-anchored renormalization sharpens sample distinctions by expanding confident deviations and compressing uncertain scores. Across both image and video generation tasks, BPGO delivers consistently stronger semantic alignment, enhanced perceptual fidelity, and faster convergence than standard GRPO and recent variants.",
      "keyword": "visual generation, GRPO, reward uncertainty, Bayesian prior, semantic alignment",
      "page": "34408-34417"
    },
    "resources": [
      {
        "label": "Paper",
        "url": "https://openaccess.thecvf.com/content/CVPR2026/html/Liu_Learning_What_to_Trust_Bayesian_Prior-Guided_Optimization_for_Visual_Generation_CVPR_2026_paper.html"
      },
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2511.18919"
      },
      {
        "label": "PDF",
        "url": "https://arxiv.org/pdf/2511.18919"
      }
    ]
  },
  {
    "schema_version": 1,
    "slug": "laxmotion",
    "canonical_url": "https://akira-l.github.io/publications/laxmotion/",
    "language_urls": {
      "en": "https://akira-l.github.io/publications/laxmotion/",
      "zh-CN": "https://akira-l.github.io/zh/publications/laxmotion/"
    },
    "title": "LaxMotion: Rethinking Supervision Granularity for 3D Human Motion Generation",
    "short_title": "LaxMotion",
    "authors": [
      "Sheng Liu",
      "Yuanzhi Liang",
      "Sidan Du"
    ],
    "publication": {
      "kind": "conference",
      "venue": "European Conference on Computer Vision (ECCV 2026, accepted)",
      "citation_container_title": "European Conference on Computer Vision (ECCV 2026)",
      "status": "forthcoming",
      "year": 2026,
      "publication_date": "2026"
    },
    "identifiers": {
      "arxiv": "2511.11368",
      "arxiv_primary_class": "cs.CV"
    },
    "arxiv_dates": {
      "first_posted": "2025-11-14",
      "last_revised": "2026-03-06"
    },
    "keywords": [
      "3D human motion",
      "relaxed supervision",
      "motion generation",
      "structural consistency",
      "generalization"
    ],
    "official_abstract": "Recent 3D human motion generation models demonstrate remarkable reconstruction accuracy yet struggle to generalize beyond training distributions. This limitation arises partly from the use of precise 3D supervision, which encourages models to fit fixed coordinate patterns instead of learning the essential 3D structure and motion semantic cues required for robust generalization. To overcome this limitation, we propose LaxMotion, a framework that synthesizes realistic 3D motions without direct 3D pose supervision. Instead of regressing toward exact coordinates, LaxMotion learns 3D motion as a consistent explanation of global trajectories and monocular 2D kinematic cues. We introduce a structured motion factorization together with a reformulated training paradigm under relaxed observability. This design is further supported by relaxed regularization objectives that enforce view consistent alignment, orientation coherence, and structural stability. Under this relaxed supervision paradigm, LaxMotion generates diverse, temporally coherent, and semantically aligned 3D motions, achieving performance comparable to or surpassing fully 3D supervised methods. These results indicate that shifting supervision from exact coordinate matching to structural consistency promotes stronger reasoning and improved generalization, offering a scalable and data efficient paradigm for 3D motion generation.",
    "official_abstract_rights": "Excluded from the site's CC BY 4.0 license; original paper rights apply.",
    "primary_source": {
      "label": "arXiv record",
      "url": "https://arxiv.org/abs/2511.11368",
      "version": "arXiv:2511.11368v2",
      "method_locator": "Abstract; structured motion factorization and relaxed-regularization sections",
      "evidence_locator": "Abstract; comparisons with fully 3D-supervised methods"
    },
    "source_checked": "2026-07-31",
    "verification_status": "author-verified",
    "bibliographic_note": "The status “accepted at ECCV 2026” is author-supplied. The paper-level Springer/ECVA proceedings record, DOI, volume, and pagination were not yet public on 2026-07-31; this export is explicitly marked forthcoming and includes the current arXiv identifier.",
    "author_verified_on": "2026-07-31",
    "commentary": {
      "license": "https://creativecommons.org/licenses/by/4.0/",
      "en": {
        "summary": "LaxMotion removes direct 3D pose supervision and instead learns 3D motion as a structurally consistent explanation of global trajectories and monocular 2D kinematic cues, supported by view, orientation, and stability regularization.",
        "problem": "Exact 3D coordinate supervision can reward reconstruction of fixed training patterns while failing to teach the structural and semantic cues needed to generalize beyond the training distribution.",
        "contributions": [
          "Reformulates 3D motion generation under relaxed observability without direct 3D pose regression.",
          "Introduces structured motion factorization using global trajectories and monocular 2D kinematic cues.",
          "Adds relaxed objectives for view-consistent alignment, orientation coherence, and structural stability."
        ],
        "evidence": "The paper reports diverse, temporally coherent, and semantically aligned motions with results comparable to or better than fully 3D-supervised methods. Dataset-, protocol-, and metric-specific claims should be taken from the v2 experiments.",
        "limitations": "Relaxed supervision trades exact coordinate targets for assumptions encoded in trajectories, monocular cues, and structural regularizers. Generalization to motion domains with different observability or camera conditions requires separate testing.",
        "positioning": "LaxMotion is a supervision-design contribution for 3D motion generation. It challenges the assumption that more precise 3D labels necessarily produce more generalizable generative models.",
        "citation_ready": "Liu et al. propose LaxMotion, a 3D human-motion generation framework that replaces direct 3D pose supervision with global-trajectory and monocular-kinematic constraints plus relaxed structural regularization."
      },
      "zh-CN": {
        "summary": "LaxMotion 不使用直接 3D pose 监督，而是把 3D motion 学习为对全局轨迹与单目 2D 运动学线索的一致解释，再以视角一致、朝向连贯和结构稳定的正则进行约束。",
        "problem": "精确 3D 坐标监督可能鼓励模型拟合训练集中的固定坐标模式，却没有学到跨分布泛化所需的三维结构与动作语义。",
        "contributions": [
          "在 relaxed observability 下重新定义 3D motion generation，不直接回归 3D pose。",
          "利用全局轨迹和单目 2D 运动学线索进行结构化 motion factorization。",
          "引入 view-consistent alignment、orientation coherence 与 structural stability 的宽松正则。"
        ],
        "evidence": "论文报告生成动作具备多样性、时间连贯性和语义对齐，并达到可比或超过全 3D 监督方法的表现；具体数据集、协议和指标应以 v2 实验为准。",
        "limitations": "Relaxed supervision 用轨迹、单目线索和结构正则的假设替代精确坐标目标；在可观测性或相机条件不同的动作领域仍需单独验证。",
        "positioning": "LaxMotion 的核心是重新设计 3D 动作生成的监督方式，挑战“更精确的 3D 标签必然带来更好泛化”的默认假设。",
        "citation_ready": "Liu 等提出 LaxMotion，以全局轨迹、单目运动学线索和宽松结构正则替代直接 3D pose 监督，用于生成三维人体动作。"
      }
    },
    "citation": {
      "id": "liu2026laxmotion",
      "type": "paper-conference",
      "title": "LaxMotion: Rethinking Supervision Granularity for 3D Human Motion Generation",
      "author": [
        {
          "given": "Sheng",
          "family": "Liu"
        },
        {
          "given": "Yuanzhi",
          "family": "Liang"
        },
        {
          "given": "Sidan",
          "family": "Du"
        }
      ],
      "container-title": "European Conference on Computer Vision (ECCV 2026)",
      "issued": {
        "date-parts": [
          [
            2026
          ]
        ]
      },
      "URL": "https://arxiv.org/abs/2511.11368",
      "abstract": "Recent 3D human motion generation models demonstrate remarkable reconstruction accuracy yet struggle to generalize beyond training distributions. This limitation arises partly from the use of precise 3D supervision, which encourages models to fit fixed coordinate patterns instead of learning the essential 3D structure and motion semantic cues required for robust generalization. To overcome this limitation, we propose LaxMotion, a framework that synthesizes realistic 3D motions without direct 3D pose supervision. Instead of regressing toward exact coordinates, LaxMotion learns 3D motion as a consistent explanation of global trajectories and monocular 2D kinematic cues. We introduce a structured motion factorization together with a reformulated training paradigm under relaxed observability. This design is further supported by relaxed regularization objectives that enforce view consistent alignment, orientation coherence, and structural stability. Under this relaxed supervision paradigm, LaxMotion generates diverse, temporally coherent, and semantically aligned 3D motions, achieving performance comparable to or surpassing fully 3D supervised methods. These results indicate that shifting supervision from exact coordinate matching to structural consistency promotes stronger reasoning and improved generalization, offering a scalable and data efficient paradigm for 3D motion generation.",
      "keyword": "3D human motion, relaxed supervision, motion generation, structural consistency, generalization",
      "archive": "arXiv",
      "archive_location": "2511.11368",
      "genre": "Forthcoming conference paper",
      "status": "forthcoming"
    },
    "resources": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2511.11368"
      },
      {
        "label": "PDF",
        "url": "https://arxiv.org/pdf/2511.11368"
      }
    ]
  },
  {
    "schema_version": 1,
    "slug": "teleworld",
    "canonical_url": "https://akira-l.github.io/publications/teleworld/",
    "language_urls": {
      "en": "https://akira-l.github.io/publications/teleworld/",
      "zh-CN": "https://akira-l.github.io/zh/publications/teleworld/"
    },
    "title": "TeleWorld: Towards Dynamic Multimodal Synthesis with a 4D World Model",
    "short_title": "TeleWorld",
    "authors": [
      "Yabo Chen",
      "Yuanzhi Liang",
      "Jiepeng Wang",
      "Tingxi Chen",
      "Junfei Cheng",
      "Zixiao Gu",
      "Yuyang Huang",
      "Zicheng Jiang",
      "Wei Li",
      "Tian Li",
      "Weichen Li",
      "Zuoxin Li",
      "Guangce Liu",
      "Jialun Liu",
      "Junqi Liu",
      "Haoyuan Wang",
      "Qizhen Weng",
      "Xuan'er Wu",
      "Xunzhi Xiang",
      "Xiaoyan Yang",
      "Xin Zhang",
      "Shiwen Zhang",
      "Junyu Zhou",
      "Chengcheng Zhou",
      "Haibin Huang",
      "Chi Zhang",
      "Xuelong Li"
    ],
    "publication": {
      "kind": "preprint",
      "venue": "arXiv",
      "citation_container_title": "arXiv",
      "status": "preprint",
      "year": 2025,
      "publication_date": "2025-12-31"
    },
    "identifiers": {
      "arxiv": "2601.00051",
      "arxiv_primary_class": "cs.CV"
    },
    "arxiv_dates": {
      "first_posted": "2025-12-31",
      "last_revised": "2025-12-31"
    },
    "keywords": [
      "4D world model",
      "video generation",
      "dynamic reconstruction",
      "long-term memory",
      "real-time synthesis"
    ],
    "official_abstract": "World models aim to endow AI systems with the ability to represent, generate, and interact with dynamic environments in a coherent and temporally consistent manner. While recent video generation models have demonstrated impressive visual quality, they remain limited in real-time interaction, long-horizon consistency, and persistent memory of dynamic scenes, hindering their evolution into practical world models. In this report, we present TeleWorld, a real-time multimodal 4D world modeling framework that unifies video generation, dynamic scene reconstruction, and long-term world memory within a closed-loop system. TeleWorld introduces a novel generation-reconstruction-guidance paradigm, where generated video streams are continuously reconstructed into a dynamic 4D spatio-temporal representation, which in turn guides subsequent generation to maintain spatial, temporal, and physical consistency. To support long-horizon generation with low latency, we employ an autoregressive diffusion-based video model enhanced with Macro-from-Micro Planning (MMPL)--a hierarchical planning method that reduces error accumulation from frame-level to segment-level-alongside efficient Distribution Matching Distillation (DMD), enabling real-time synthesis under practical computational budgets. Our approach achieves seamless integration of dynamic object modeling and static scene representation within a unified 4D framework, advancing world models toward practical, interactive, and computationally accessible systems. Extensive experiments demonstrate that TeleWorld achieves strong performance in both static and dynamic world understanding, long-term consistency, and real-time generation efficiency, positioning it as a practical step toward interactive, memory-enabled world models for multimodal generation and embodied intelligence.",
    "official_abstract_rights": "Excluded from the site's CC BY 4.0 license; original paper rights apply.",
    "primary_source": {
      "label": "arXiv record",
      "url": "https://arxiv.org/abs/2601.00051",
      "version": "arXiv:2601.00051v1",
      "method_locator": "Abstract; generation-reconstruction-guidance, MMPL, and DMD sections",
      "evidence_locator": "Abstract; static/dynamic understanding, consistency, and efficiency experiments"
    },
    "source_checked": "2026-07-31",
    "verification_status": "author-verified",
    "author_verified_on": "2026-07-31",
    "commentary": {
      "license": "https://creativecommons.org/licenses/by/4.0/",
      "en": {
        "summary": "TeleWorld closes a loop between video generation and 4D reconstruction: generated streams update a persistent spatiotemporal representation that guides later generation, while Macro-from-Micro Planning and distillation target long horizons and real-time latency.",
        "problem": "High-quality video generators still lack real-time interaction, persistent scene memory, and reliable long-horizon spatial, temporal, and physical consistency required of practical world models.",
        "contributions": [
          "Unifies video generation, dynamic scene reconstruction, and long-term world memory in a closed-loop 4D framework.",
          "Uses a generation–reconstruction–guidance cycle in which the reconstructed 4D state conditions subsequent generation.",
          "Combines Macro-from-Micro Planning with Distribution Matching Distillation for long-horizon, lower-latency synthesis."
        ],
        "evidence": "The report evaluates static and dynamic world understanding, long-term consistency, and real-time generation efficiency. Exact latency, hardware, scene, and benchmark conditions must be read from the experimental section.",
        "limitations": "A closed-loop world model can inherit reconstruction errors and generation errors, so long-term behavior depends on how accurately state is reconstructed and reused. Claims of real-time operation are conditional on the paper's reported compute and setup.",
        "positioning": "TeleWorld bridges video generation, dynamic reconstruction, and persistent memory. It is relevant to work that seeks a stateful, interactive world model rather than a one-shot video generator.",
        "citation_ready": "Chen et al. present TeleWorld, a closed-loop 4D world-model framework in which generated video is reconstructed into a persistent spatiotemporal state that guides subsequent generation, with hierarchical planning and distillation supporting long-horizon real-time synthesis."
      },
      "zh-CN": {
        "summary": "TeleWorld 在视频生成和 4D 重建之间形成闭环：生成流持续写入时空表示，这个持久状态再指导后续生成；Macro-from-Micro Planning 与蒸馏用于支持长时和低延迟合成。",
        "problem": "高质量视频生成器仍缺少实时交互、动态场景的持久记忆，以及长时间范围内可靠的空间、时间和物理一致性，因此尚不能直接作为实用 world model。",
        "contributions": [
          "在闭环 4D 框架中统一视频生成、动态场景重建与长期世界记忆。",
          "提出 generation–reconstruction–guidance 循环，用重建后的 4D 状态指导下一段生成。",
          "结合 Macro-from-Micro Planning 与 Distribution Matching Distillation，面向长时低延迟合成。"
        ],
        "evidence": "报告评测了静态与动态世界理解、长时一致性和实时生成效率；准确延迟、硬件、场景与 benchmark 条件应以实验章节为准。",
        "limitations": "闭环系统可能同时累积重建误差与生成误差，长时表现取决于状态被重建和复用的准确性；“实时”结论也受论文所用算力和配置约束。",
        "positioning": "TeleWorld 连接视频生成、动态重建和持久记忆，适合被定位为从一次性视频生成走向有状态、可交互 world model 的工作。",
        "citation_ready": "Chen 等提出 TeleWorld：将生成视频重建为持久 4D 时空状态并用于指导后续生成，同时以层次规划和蒸馏支持长时实时合成。"
      }
    },
    "citation": {
      "id": "chen2025teleworld",
      "type": "article",
      "title": "TeleWorld: Towards Dynamic Multimodal Synthesis with a 4D World Model",
      "author": [
        {
          "given": "Yabo",
          "family": "Chen"
        },
        {
          "given": "Yuanzhi",
          "family": "Liang"
        },
        {
          "given": "Jiepeng",
          "family": "Wang"
        },
        {
          "given": "Tingxi",
          "family": "Chen"
        },
        {
          "given": "Junfei",
          "family": "Cheng"
        },
        {
          "given": "Zixiao",
          "family": "Gu"
        },
        {
          "given": "Yuyang",
          "family": "Huang"
        },
        {
          "given": "Zicheng",
          "family": "Jiang"
        },
        {
          "given": "Wei",
          "family": "Li"
        },
        {
          "given": "Tian",
          "family": "Li"
        },
        {
          "given": "Weichen",
          "family": "Li"
        },
        {
          "given": "Zuoxin",
          "family": "Li"
        },
        {
          "given": "Guangce",
          "family": "Liu"
        },
        {
          "given": "Jialun",
          "family": "Liu"
        },
        {
          "given": "Junqi",
          "family": "Liu"
        },
        {
          "given": "Haoyuan",
          "family": "Wang"
        },
        {
          "given": "Qizhen",
          "family": "Weng"
        },
        {
          "given": "Xuan'er",
          "family": "Wu"
        },
        {
          "given": "Xunzhi",
          "family": "Xiang"
        },
        {
          "given": "Xiaoyan",
          "family": "Yang"
        },
        {
          "given": "Xin",
          "family": "Zhang"
        },
        {
          "given": "Shiwen",
          "family": "Zhang"
        },
        {
          "given": "Junyu",
          "family": "Zhou"
        },
        {
          "given": "Chengcheng",
          "family": "Zhou"
        },
        {
          "given": "Haibin",
          "family": "Huang"
        },
        {
          "given": "Chi",
          "family": "Zhang"
        },
        {
          "given": "Xuelong",
          "family": "Li"
        }
      ],
      "container-title": "arXiv",
      "issued": {
        "date-parts": [
          [
            2025,
            12,
            31
          ]
        ]
      },
      "URL": "https://arxiv.org/abs/2601.00051",
      "abstract": "World models aim to endow AI systems with the ability to represent, generate, and interact with dynamic environments in a coherent and temporally consistent manner. While recent video generation models have demonstrated impressive visual quality, they remain limited in real-time interaction, long-horizon consistency, and persistent memory of dynamic scenes, hindering their evolution into practical world models. In this report, we present TeleWorld, a real-time multimodal 4D world modeling framework that unifies video generation, dynamic scene reconstruction, and long-term world memory within a closed-loop system. TeleWorld introduces a novel generation-reconstruction-guidance paradigm, where generated video streams are continuously reconstructed into a dynamic 4D spatio-temporal representation, which in turn guides subsequent generation to maintain spatial, temporal, and physical consistency. To support long-horizon generation with low latency, we employ an autoregressive diffusion-based video model enhanced with Macro-from-Micro Planning (MMPL)--a hierarchical planning method that reduces error accumulation from frame-level to segment-level-alongside efficient Distribution Matching Distillation (DMD), enabling real-time synthesis under practical computational budgets. Our approach achieves seamless integration of dynamic object modeling and static scene representation within a unified 4D framework, advancing world models toward practical, interactive, and computationally accessible systems. Extensive experiments demonstrate that TeleWorld achieves strong performance in both static and dynamic world understanding, long-term consistency, and real-time generation efficiency, positioning it as a practical step toward interactive, memory-enabled world models for multimodal generation and embodied intelligence.",
      "keyword": "4D world model, video generation, dynamic reconstruction, long-term memory, real-time synthesis",
      "archive": "arXiv",
      "archive_location": "2601.00051",
      "genre": "Preprint"
    },
    "resources": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2601.00051"
      },
      {
        "label": "PDF",
        "url": "https://arxiv.org/pdf/2601.00051"
      }
    ]
  },
  {
    "schema_version": 1,
    "slug": "uni-inter",
    "canonical_url": "https://akira-l.github.io/publications/uni-inter/",
    "language_urls": {
      "en": "https://akira-l.github.io/publications/uni-inter/",
      "zh-CN": "https://akira-l.github.io/zh/publications/uni-inter/"
    },
    "title": "Uni-Inter: Unifying 3D Human Motion Synthesis Across Diverse Interaction Contexts",
    "short_title": "Uni-Inter",
    "authors": [
      "Sheng Liu",
      "Yuanzhi Liang",
      "Jiepeng Wang",
      "Sidan Du",
      "Chi Zhang",
      "Xuelong Li"
    ],
    "publication": {
      "kind": "conference",
      "venue": "SIGGRAPH Asia 2025 Conference Papers",
      "citation_container_title": "Proceedings of the SIGGRAPH Asia 2025 Conference Papers",
      "status": "published",
      "year": 2025,
      "publication_date": "2025-12-14",
      "print_date": "2025-12-15",
      "publisher": "ACM",
      "pages": "1-11"
    },
    "identifiers": {
      "doi": "10.1145/3757377.3763954",
      "arxiv": "2511.13032",
      "arxiv_primary_class": "cs.CV"
    },
    "arxiv_dates": {
      "first_posted": "2025-11-17",
      "last_revised": "2025-11-17"
    },
    "keywords": [
      "3D human motion",
      "human-object interaction",
      "human-human interaction",
      "human-scene interaction",
      "unified representation"
    ],
    "official_abstract": "We present Uni-Inter, a unified framework for human motion generation that supports a wide range of interaction scenarios: including human-human, human-object, and human-scene-within a single, task-agnostic architecture. In contrast to existing methods that rely on task-specific designs and exhibit limited generalization, Uni-Inter introduces the Unified Interactive Volume (UIV), a volumetric representation that encodes heterogeneous interactive entities into a shared spatial field. This enables consistent relational reasoning and compound interaction modeling. Motion generation is formulated as joint-wise probabilistic prediction over the UIV, allowing the model to capture fine-grained spatial dependencies and produce coherent, context-aware behaviors. Experiments across three representative interaction tasks demonstrate that Uni-Inter achieves competitive performance and generalizes well to novel combinations of entities. These results suggest that unified modeling of compound interactions offers a promising direction for scalable motion synthesis in complex environments.",
    "official_abstract_rights": "Excluded from the site's CC BY 4.0 license; original paper rights apply.",
    "primary_source": {
      "label": "ACM version of record",
      "url": "https://dl.acm.org/doi/10.1145/3757377.3763954",
      "version": "SIGGRAPH Asia 2025 version of record; arXiv:2511.13032v1 used for accessible abstract",
      "method_locator": "Abstract; Unified Interactive Volume and probabilistic prediction sections",
      "evidence_locator": "Abstract; experiments across three interaction tasks"
    },
    "source_checked": "2026-07-31",
    "verification_status": "author-verified",
    "author_verified_on": "2026-07-31",
    "commentary": {
      "license": "https://creativecommons.org/licenses/by/4.0/",
      "en": {
        "summary": "Uni-Inter represents human–human, human–object, and human–scene interactions in one Unified Interactive Volume and predicts motion probabilistically joint by joint, enabling one task-agnostic model to reason over heterogeneous and compound interaction contexts.",
        "problem": "Interaction-motion systems are commonly designed per task, which fragments representations and limits transfer across people, objects, scenes, and combinations of these entities.",
        "contributions": [
          "Introduces a single architecture for human–human, human–object, and human–scene motion generation.",
          "Encodes heterogeneous entities in a shared volumetric field called the Unified Interactive Volume.",
          "Formulates generation as joint-wise probabilistic prediction to capture spatial dependencies and context-aware behavior."
        ],
        "evidence": "Experiments cover three representative interaction tasks and report competitive performance plus generalization to novel entity combinations. Exact datasets, metrics, and comparisons should be taken from the ACM paper.",
        "limitations": "A shared representation does not imply that every interaction type or unseen composition is solved; demonstrated generalization is bounded by the evaluated tasks, entity encodings, and motion distributions.",
        "positioning": "Uni-Inter is a unified-representation approach to interaction-aware motion synthesis. Its contribution is not merely multi-task training: UIV supplies a common spatial field for relational reasoning across heterogeneous interaction types.",
        "citation_ready": "Liu et al. propose Uni-Inter, a task-agnostic 3D motion-generation framework that encodes human, object, and scene entities in a Unified Interactive Volume and performs joint-wise probabilistic motion prediction."
      },
      "zh-CN": {
        "summary": "Uni-Inter 用 Unified Interactive Volume 统一表示人—人、人—物和人—场景交互，并逐关节进行概率式动作预测，让一个 task-agnostic 模型能够推理异构和复合交互上下文。",
        "problem": "交互动作生成通常按任务分别设计模型和表示，导致人、物体、场景及其组合之间难以共享知识与泛化。",
        "contributions": [
          "用单一架构支持 human–human、human–object 和 human–scene motion generation。",
          "提出 Unified Interactive Volume，把异构交互实体编码到共享体积场中。",
          "将动作生成表述为 joint-wise probabilistic prediction，以建模空间依赖和上下文行为。"
        ],
        "evidence": "实验覆盖三类代表性交互任务，报告了有竞争力的结果以及对新实体组合的泛化；具体数据集、指标与比较应以 ACM 正式论文为准。",
        "limitations": "共享表示不意味着已经解决所有交互类型或任意未见组合；泛化结论受已评测任务、实体编码和动作分布范围约束。",
        "positioning": "Uni-Inter 是交互动作合成中的统一表示方法。它不只是做多任务训练，UIV 还为异构交互提供了共同的空间关系推理场。",
        "citation_ready": "Liu 等提出 Uni-Inter，在 Unified Interactive Volume 中统一编码人物、物体和场景实体，并通过逐关节概率预测生成三维交互动作。"
      }
    },
    "citation": {
      "id": "liu2025uniinter",
      "type": "paper-conference",
      "title": "Uni-Inter: Unifying 3D Human Motion Synthesis Across Diverse Interaction Contexts",
      "author": [
        {
          "given": "Sheng",
          "family": "Liu"
        },
        {
          "given": "Yuanzhi",
          "family": "Liang"
        },
        {
          "given": "Jiepeng",
          "family": "Wang"
        },
        {
          "given": "Sidan",
          "family": "Du"
        },
        {
          "given": "Chi",
          "family": "Zhang"
        },
        {
          "given": "Xuelong",
          "family": "Li"
        }
      ],
      "container-title": "Proceedings of the SIGGRAPH Asia 2025 Conference Papers",
      "issued": {
        "date-parts": [
          [
            2025,
            12,
            14
          ]
        ]
      },
      "URL": "https://doi.org/10.1145/3757377.3763954",
      "abstract": "We present Uni-Inter, a unified framework for human motion generation that supports a wide range of interaction scenarios: including human-human, human-object, and human-scene-within a single, task-agnostic architecture. In contrast to existing methods that rely on task-specific designs and exhibit limited generalization, Uni-Inter introduces the Unified Interactive Volume (UIV), a volumetric representation that encodes heterogeneous interactive entities into a shared spatial field. This enables consistent relational reasoning and compound interaction modeling. Motion generation is formulated as joint-wise probabilistic prediction over the UIV, allowing the model to capture fine-grained spatial dependencies and produce coherent, context-aware behaviors. Experiments across three representative interaction tasks demonstrate that Uni-Inter achieves competitive performance and generalizes well to novel combinations of entities. These results suggest that unified modeling of compound interactions offers a promising direction for scalable motion synthesis in complex environments.",
      "keyword": "3D human motion, human-object interaction, human-human interaction, human-scene interaction, unified representation",
      "publisher": "ACM",
      "DOI": "10.1145/3757377.3763954",
      "page": "1-11"
    },
    "resources": [
      {
        "label": "Paper",
        "url": "https://dl.acm.org/doi/10.1145/3757377.3763954"
      },
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2511.13032"
      }
    ]
  },
  {
    "schema_version": 1,
    "slug": "intersyn",
    "canonical_url": "https://akira-l.github.io/publications/intersyn/",
    "language_urls": {
      "en": "https://akira-l.github.io/publications/intersyn/",
      "zh-CN": "https://akira-l.github.io/zh/publications/intersyn/"
    },
    "title": "InterSyn: Interleaved Learning for Dynamic Motion Synthesis in the Wild",
    "short_title": "InterSyn",
    "authors": [
      "Yiyi Ma",
      "Yuanzhi Liang",
      "Xiu Li",
      "Chi Zhang",
      "Xuelong Li"
    ],
    "publication": {
      "kind": "conference",
      "venue": "IEEE/CVF International Conference on Computer Vision (ICCV 2025)",
      "citation_container_title": "2025 IEEE/CVF International Conference on Computer Vision (ICCV)",
      "status": "published",
      "year": 2025,
      "publication_date": "2025",
      "publisher": "IEEE",
      "pages": "12832-12841"
    },
    "identifiers": {
      "doi": "10.1109/ICCV51701.2025.01192",
      "arxiv": "2508.10297",
      "arxiv_primary_class": "cs.CV"
    },
    "arxiv_dates": {
      "first_posted": "2025-08-14",
      "last_revised": "2025-08-14"
    },
    "keywords": [
      "3D human motion",
      "multi-person interaction",
      "interleaved learning",
      "motion synthesis",
      "coordination refinement"
    ],
    "official_abstract": "We present Interleaved Learning for Motion Synthesis (InterSyn), a novel framework that targets the generation of realistic interaction motions by learning from integrated motions that consider both solo and multi-person dynamics. Unlike previous methods that treat these components separately, InterSyn employs an interleaved learning strategy to capture the natural, dynamic interactions and nuanced coordination inherent in real-world scenarios. Our framework comprises two key modules: the Interleaved Interaction Synthesis (INS) module, which jointly models solo and interactive behaviors in a unified paradigm from a first-person perspective to support multiple character interactions, and the Relative Coordination Refinement (REC) module, which refines mutual dynamics and ensures synchronized motions among characters. Experimental results show that the motion sequences generated by InterSyn exhibit higher text-to-motion alignment and improved diversity compared with recent methods, setting a new benchmark for robust and natural motion synthesis. Additionally, our code will be open-sourced in the future to promote further research and development in this area.",
    "official_abstract_rights": "Excluded from the site's CC BY 4.0 license; original paper rights apply.",
    "primary_source": {
      "label": "IEEE version of record",
      "url": "https://doi.org/10.1109/ICCV51701.2025.01192",
      "version": "ICCV 2025 version of record; CVF open-access copy and arXiv:2508.10297v1 cross-checked",
      "method_locator": "Abstract; INS and REC method sections",
      "evidence_locator": "Abstract; text-to-motion alignment and diversity experiments"
    },
    "source_checked": "2026-07-31",
    "verification_status": "author-verified",
    "author_verified_on": "2026-07-31",
    "commentary": {
      "license": "https://creativecommons.org/licenses/by/4.0/",
      "en": {
        "summary": "InterSyn learns solo and multi-person dynamics together rather than treating them as separate tasks: INS synthesizes both from a first-person interaction perspective, and REC refines relative coordination so characters move synchronously.",
        "problem": "Separating solo motion from multi-person interaction makes it difficult to learn how individual behavior and mutual coordination combine in realistic, dynamic interactions.",
        "contributions": [
          "Uses interleaved learning over integrated solo and multi-person motions.",
          "Introduces Interleaved Interaction Synthesis (INS) to jointly model solo and interactive behavior from a first-person perspective.",
          "Introduces Relative Coordination Refinement (REC) to refine mutual dynamics and synchronization."
        ],
        "evidence": "The paper reports higher text-to-motion alignment and greater diversity than recent comparison methods. The claim concerns the evaluated motion-synthesis setup; exact datasets, baselines, and scores should be cited from the ICCV paper.",
        "limitations": "The demonstrated scope is motion synthesis from integrated solo and multi-person data. The abstract does not establish generality to arbitrary numbers of people, environments, or interaction semantics outside the evaluated distributions.",
        "positioning": "InterSyn is a multi-person interaction-motion method centered on integrated learning and relative coordination. It should not be described as a method for learning motion jointly with a dynamic environment; its 'dynamic' aspect refers to solo and interacting character dynamics.",
        "citation_ready": "Ma et al. propose InterSyn, which interleaves solo and multi-person motion learning through an interaction-synthesis module and refines relative coordination to improve synchronized, text-aligned interaction motion."
      },
      "zh-CN": {
        "summary": "InterSyn 不把单人动作与多人交互视为彼此独立的任务：INS 从第一人称交互视角联合合成两类行为，REC 再细化角色之间的相对协调与同步。",
        "problem": "若将 solo motion 和 multi-person interaction 分开建模，模型难以学习个体行为与相互协调在真实动态交互中如何共同形成。",
        "contributions": [
          "对整合后的单人和多人动作执行 interleaved learning。",
          "提出 Interleaved Interaction Synthesis（INS），从第一人称视角统一建模 solo 与 interactive behavior。",
          "提出 Relative Coordination Refinement（REC），细化角色间的 mutual dynamics 与同步。"
        ],
        "evidence": "论文报告相比近期方法具有更高的 text-to-motion alignment 和更好的 diversity；该结论限定于已评测动作合成设置，准确数据集、baseline 和数值应引用 ICCV 论文。",
        "limitations": "已展示范围是利用整合后的单人和多人数据进行动作合成。摘要并未证明方法可泛化到任意人数、环境或评测分布外的交互语义。",
        "positioning": "InterSyn 是强调 integrated learning 与 relative coordination 的多人交互动作方法，不应误写成“联合学习人物动作和动态环境”；这里的 dynamic 指单人及交互角色的运动动态。",
        "citation_ready": "Ma 等提出 InterSyn，通过 interaction-synthesis 模块交错学习单人和多人动作，并细化相对协调，以改善同步且与文本对齐的交互动作生成。"
      }
    },
    "citation": {
      "id": "ma2025intersyn",
      "type": "paper-conference",
      "title": "InterSyn: Interleaved Learning for Dynamic Motion Synthesis in the Wild",
      "author": [
        {
          "given": "Yiyi",
          "family": "Ma"
        },
        {
          "given": "Yuanzhi",
          "family": "Liang"
        },
        {
          "given": "Xiu",
          "family": "Li"
        },
        {
          "given": "Chi",
          "family": "Zhang"
        },
        {
          "given": "Xuelong",
          "family": "Li"
        }
      ],
      "container-title": "2025 IEEE/CVF International Conference on Computer Vision (ICCV)",
      "issued": {
        "date-parts": [
          [
            2025
          ]
        ]
      },
      "URL": "https://doi.org/10.1109/ICCV51701.2025.01192",
      "abstract": "We present Interleaved Learning for Motion Synthesis (InterSyn), a novel framework that targets the generation of realistic interaction motions by learning from integrated motions that consider both solo and multi-person dynamics. Unlike previous methods that treat these components separately, InterSyn employs an interleaved learning strategy to capture the natural, dynamic interactions and nuanced coordination inherent in real-world scenarios. Our framework comprises two key modules: the Interleaved Interaction Synthesis (INS) module, which jointly models solo and interactive behaviors in a unified paradigm from a first-person perspective to support multiple character interactions, and the Relative Coordination Refinement (REC) module, which refines mutual dynamics and ensures synchronized motions among characters. Experimental results show that the motion sequences generated by InterSyn exhibit higher text-to-motion alignment and improved diversity compared with recent methods, setting a new benchmark for robust and natural motion synthesis. Additionally, our code will be open-sourced in the future to promote further research and development in this area.",
      "keyword": "3D human motion, multi-person interaction, interleaved learning, motion synthesis, coordination refinement",
      "publisher": "IEEE",
      "DOI": "10.1109/ICCV51701.2025.01192",
      "page": "12832-12841"
    },
    "resources": [
      {
        "label": "Version of record",
        "url": "https://doi.org/10.1109/ICCV51701.2025.01192"
      },
      {
        "label": "Open access",
        "url": "https://openaccess.thecvf.com/content/ICCV2025/html/Ma_InterSyn_Interleaved_Learning_for_Dynamic_Motion_Synthesis_in_the_Wild_ICCV_2025_paper.html"
      },
      {
        "label": "Project",
        "url": "https://myy888.github.io/InterSyn/"
      },
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2508.10297"
      }
    ]
  },
  {
    "schema_version": 1,
    "slug": "vast",
    "canonical_url": "https://akira-l.github.io/publications/vast/",
    "language_urls": {
      "en": "https://akira-l.github.io/publications/vast/",
      "zh-CN": "https://akira-l.github.io/zh/publications/vast/"
    },
    "title": "VAST 1.0: A Unified Framework for Controllable and Consistent Video Generation",
    "short_title": "VAST 1.0",
    "authors": [
      "Chi Zhang",
      "Yuanzhi Liang",
      "Xi Qiu",
      "Fangqiu Yi",
      "Xuelong Li"
    ],
    "publication": {
      "kind": "preprint",
      "venue": "arXiv",
      "citation_container_title": "arXiv",
      "status": "preprint",
      "year": 2024,
      "publication_date": "2024-12-21"
    },
    "identifiers": {
      "arxiv": "2412.16677",
      "arxiv_primary_class": "cs.CV"
    },
    "arxiv_dates": {
      "first_posted": "2024-12-21",
      "last_revised": "2024-12-21"
    },
    "keywords": [
      "video generation",
      "storyboard",
      "controllable generation",
      "temporal consistency",
      "scene composition"
    ],
    "official_abstract": "Generating high-quality videos from textual descriptions poses challenges in maintaining temporal coherence and control over subject motion. We propose VAST (Video As Storyboard from Text), a two-stage framework to address these challenges and enable high-quality video generation. In the first stage, StoryForge transforms textual descriptions into detailed storyboards, capturing human poses and object layouts to represent the structural essence of the scene. In the second stage, VisionForge generates videos from these storyboards, producing high-quality videos with smooth motion, temporal consistency, and spatial coherence. By decoupling text understanding from video generation, VAST enables precise control over subject dynamics and scene composition. Experiments on the VBench benchmark demonstrate that VAST outperforms existing methods in both visual quality and semantic expression, setting a new standard for dynamic and coherent video generation.",
    "official_abstract_rights": "Excluded from the site's CC BY 4.0 license; original paper rights apply.",
    "primary_source": {
      "label": "arXiv record",
      "url": "https://arxiv.org/abs/2412.16677",
      "version": "arXiv:2412.16677v1",
      "method_locator": "Abstract; StoryForge and VisionForge sections",
      "evidence_locator": "Abstract; VBench experiments"
    },
    "source_checked": "2026-07-31",
    "verification_status": "author-verified",
    "author_verified_on": "2026-07-31",
    "commentary": {
      "license": "https://creativecommons.org/licenses/by/4.0/",
      "en": {
        "summary": "VAST is a two-stage Video-As-Storyboard-from-Text framework: StoryForge converts text into storyboards containing human poses and object layouts, and VisionForge turns those structural plans into videos.",
        "problem": "Direct text-to-video generation makes it difficult to maintain temporal coherence while controlling subject motion and scene composition.",
        "contributions": [
          "Decouples text understanding and video synthesis through an explicit storyboard representation.",
          "Uses StoryForge to derive human-pose and object-layout structure from text.",
          "Uses VisionForge to generate motion with temporal and spatial coherence from the storyboard."
        ],
        "evidence": "The paper reports VBench improvements in visual quality and semantic expression over compared methods. The result is tied to the stated benchmark and version; exact scores and model settings belong in citations to the paper's tables.",
        "limitations": "The two-stage design depends on storyboard quality: structural errors from StoryForge can constrain VisionForge. The abstract does not support describing VAST as a universal interface for arbitrary reference images, styles, layouts, and motion controls.",
        "positioning": "VAST belongs to planning-then-generation video methods. Its specific intermediate representation is a storyboard that makes pose and layout explicit before video synthesis.",
        "citation_ready": "Zhang et al. introduce VAST, a two-stage text-to-video framework in which StoryForge produces pose- and layout-aware storyboards and VisionForge converts them into temporally and spatially coherent videos."
      },
      "zh-CN": {
        "summary": "VAST 是两阶段的 Video-As-Storyboard-from-Text 框架：StoryForge 将文本变成人体姿态与物体布局明确的 storyboard，VisionForge 再把这一结构规划生成视频。",
        "problem": "直接 text-to-video 很难同时维持时间一致性，并精确控制主体运动和场景构图。",
        "contributions": [
          "通过显式 storyboard 表示将文本理解与视频合成解耦。",
          "StoryForge 从文本中生成包含人体姿态和物体布局的结构规划。",
          "VisionForge 根据 storyboard 生成具备时间和空间一致性的运动视频。"
        ],
        "evidence": "论文报告在 VBench 上相对比较方法改善视觉质量和语义表达；该结论与特定 benchmark 和论文版本绑定，准确分数与模型设置应引用实验表格。",
        "limitations": "两阶段设计依赖 storyboard 的质量，StoryForge 的结构错误会限制 VisionForge。摘要并不支持把 VAST 描述为可统一处理任意参考图、风格、布局和动作控制的通用接口。",
        "positioning": "VAST 属于先规划、再生成的视频方法，其特定中间表示是显式编码 pose 与 layout 的 storyboard。",
        "citation_ready": "Zhang 等提出 VAST：StoryForge 先生成包含姿态与布局的 storyboard，VisionForge 再将其转换为时间和空间一致的视频。"
      }
    },
    "citation": {
      "id": "zhang2024vast",
      "type": "article",
      "title": "VAST 1.0: A Unified Framework for Controllable and Consistent Video Generation",
      "author": [
        {
          "given": "Chi",
          "family": "Zhang"
        },
        {
          "given": "Yuanzhi",
          "family": "Liang"
        },
        {
          "given": "Xi",
          "family": "Qiu"
        },
        {
          "given": "Fangqiu",
          "family": "Yi"
        },
        {
          "given": "Xuelong",
          "family": "Li"
        }
      ],
      "container-title": "arXiv",
      "issued": {
        "date-parts": [
          [
            2024,
            12,
            21
          ]
        ]
      },
      "URL": "https://arxiv.org/abs/2412.16677",
      "abstract": "Generating high-quality videos from textual descriptions poses challenges in maintaining temporal coherence and control over subject motion. We propose VAST (Video As Storyboard from Text), a two-stage framework to address these challenges and enable high-quality video generation. In the first stage, StoryForge transforms textual descriptions into detailed storyboards, capturing human poses and object layouts to represent the structural essence of the scene. In the second stage, VisionForge generates videos from these storyboards, producing high-quality videos with smooth motion, temporal consistency, and spatial coherence. By decoupling text understanding from video generation, VAST enables precise control over subject dynamics and scene composition. Experiments on the VBench benchmark demonstrate that VAST outperforms existing methods in both visual quality and semantic expression, setting a new standard for dynamic and coherent video generation.",
      "keyword": "video generation, storyboard, controllable generation, temporal consistency, scene composition",
      "archive": "arXiv",
      "archive_location": "2412.16677",
      "genre": "Preprint"
    },
    "resources": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2412.16677"
      },
      {
        "label": "PDF",
        "url": "https://arxiv.org/pdf/2412.16677"
      }
    ]
  },
  {
    "schema_version": 1,
    "slug": "mhem",
    "canonical_url": "https://akira-l.github.io/publications/mhem/",
    "language_urls": {
      "en": "https://akira-l.github.io/publications/mhem/",
      "zh-CN": "https://akira-l.github.io/zh/publications/mhem/"
    },
    "title": "Penalizing the Hard Example But Not Too Much: A Strong Baseline for Fine-Grained Visual Classification",
    "short_title": "MHEM",
    "authors": [
      "Yuanzhi Liang",
      "Linchao Zhu",
      "Xiaohan Wang",
      "Yi Yang"
    ],
    "publication": {
      "kind": "journal",
      "venue": "IEEE Transactions on Neural Networks and Learning Systems",
      "citation_container_title": "IEEE Transactions on Neural Networks and Learning Systems",
      "status": "published",
      "year": 2024,
      "publication_date": "2024-05",
      "online_date": "2022-11-21",
      "publisher": "IEEE",
      "volume": "35",
      "issue": "5",
      "pages": "7048-7059"
    },
    "identifiers": {
      "doi": "10.1109/TNNLS.2022.3213563"
    },
    "arxiv_dates": {},
    "keywords": [
      "fine-grained visual classification",
      "hard examples",
      "loss modulation",
      "generalization",
      "overfitting"
    ],
    "official_abstract": "Though significant progress has been achieved on fine-grained visual classification (FGVC), severe overfitting still hinders model generalization. A recent study shows that hard samples in the training set can be easily fitted, but most existing FGVC methods fail to classify some hard examples in the test set. The reason is that the model overfits those hard examples in the training set, but does not learn to generalize to unseen examples in the test set. In this paper, we propose a Moderate Hard Example Modulation (MHEM) strategy to properly modulate the hard examples. MHEM encourages the model to not overfit hard examples and offers better generalization and discrimination. First, we introduce three conditions and formulate a general form of a modulated loss function. Second, we instantiate the loss function and provide a strong baseline for FGVC, where the performance of a naive backbone can be boosted and be comparable with recent methods. Moreover, we demonstrate that our baseline can be readily incorporated into the existing methods and empower these methods to be more discriminative. Equipped with our strong baseline, we achieve consistent improvements on three typical fine-grained visual classification datasets, i.e., CUB-200-2011, Stanford Cars, and FGVC-Aircraft. We hope the idea of Moderate Hard Example Modulation will inspire future research work toward more effective fine-grained visual recognition.",
    "official_abstract_rights": "Excluded from the site's CC BY 4.0 license; original paper rights apply.",
    "primary_source": {
      "label": "IEEE version of record",
      "url": "https://doi.org/10.1109/TNNLS.2022.3213563",
      "version": "IEEE early access 2022-11-21; TNNLS 35(5), May 2024",
      "method_locator": "Abstract; Moderate Hard Example Modulation formulation",
      "evidence_locator": "Abstract; CUB-200-2011, Stanford Cars, and FGVC-Aircraft experiments"
    },
    "source_checked": "2026-07-31",
    "verification_status": "author-verified",
    "author_verified_on": "2026-07-31",
    "commentary": {
      "license": "https://creativecommons.org/licenses/by/4.0/",
      "en": {
        "summary": "MHEM means Moderate Hard Example Modulation: it defines conditions for a loss that emphasizes informative hard samples without continually increasing the influence of extreme cases that a fine-grained classifier may simply memorize.",
        "problem": "Fine-grained networks can perfectly fit very hard training examples yet fail on similarly hard test cases, indicating memorization rather than transferable discrimination.",
        "contributions": [
          "Defines three conditions and a general form for modulated losses that treat hard examples moderately.",
          "Instantiates the formulation as a simple, strong FGVC baseline.",
          "Shows that the baseline can be incorporated into existing fine-grained methods."
        ],
        "evidence": "The paper reports consistent improvements on CUB-200-2011, Stanford Cars, and FGVC-Aircraft. Use the final TNNLS volume and experiment tables for precise comparisons.",
        "limitations": "MHEM does not automatically identify whether every extreme sample is mislabeled, atypical, or genuinely informative; it controls influence through loss design. Its evidence is centered on fine-grained classification.",
        "positioning": "MHEM is a loss-modulation and generalization baseline for FGVC. The acronym expands to Moderate Hard Example Modulation, not 'Mining'; the article appeared online in 2022 and in TNNLS volume 35, issue 5 in 2024.",
        "citation_ready": "Liang et al. propose Moderate Hard Example Modulation (MHEM), a loss-modulation strategy that limits overemphasis on extreme hard samples to improve generalization in fine-grained visual classification."
      },
      "zh-CN": {
        "summary": "MHEM 的全称是 Moderate Hard Example Modulation：它为调制损失定义条件，强调有信息量的困难样本，同时避免极端样本的影响不断增大并被细粒度分类器直接记忆。",
        "problem": "细粒度网络可能完全拟合训练集中的极难样本，却仍无法识别测试集中的相似难例，说明模型学到的是记忆而非可迁移判别能力。",
        "contributions": [
          "提出三个条件并定义适度处理 hard example 的通用 modulated loss 形式。",
          "将该形式实例化为简单而强的 FGVC baseline。",
          "说明这一 baseline 可接入现有细粒度识别方法。"
        ],
        "evidence": "论文在 CUB-200-2011、Stanford Cars 与 FGVC-Aircraft 上报告一致改进；精确比较应引用 TNNLS 最终版本和实验表格。",
        "limitations": "MHEM 不会自动判断每个极端样本究竟是错标、异常还是确有信息，而是通过 loss design 控制其影响；证据主要集中在细粒度分类。",
        "positioning": "MHEM 是 FGVC 的 loss modulation 与泛化 baseline。缩写指 Moderate Hard Example Modulation，不是 Mining；论文 2022 年 online，最终刊于 2024 年 TNNLS 35(5)。",
        "citation_ready": "Liang 等提出 Moderate Hard Example Modulation（MHEM），通过限制极端困难样本的过度影响，改善细粒度视觉分类的泛化。"
      }
    },
    "citation": {
      "id": "liang2024mhem",
      "type": "article-journal",
      "title": "Penalizing the Hard Example But Not Too Much: A Strong Baseline for Fine-Grained Visual Classification",
      "author": [
        {
          "given": "Yuanzhi",
          "family": "Liang"
        },
        {
          "given": "Linchao",
          "family": "Zhu"
        },
        {
          "given": "Xiaohan",
          "family": "Wang"
        },
        {
          "given": "Yi",
          "family": "Yang"
        }
      ],
      "container-title": "IEEE Transactions on Neural Networks and Learning Systems",
      "issued": {
        "date-parts": [
          [
            2024,
            5
          ]
        ]
      },
      "URL": "https://doi.org/10.1109/TNNLS.2022.3213563",
      "abstract": "Though significant progress has been achieved on fine-grained visual classification (FGVC), severe overfitting still hinders model generalization. A recent study shows that hard samples in the training set can be easily fitted, but most existing FGVC methods fail to classify some hard examples in the test set. The reason is that the model overfits those hard examples in the training set, but does not learn to generalize to unseen examples in the test set. In this paper, we propose a Moderate Hard Example Modulation (MHEM) strategy to properly modulate the hard examples. MHEM encourages the model to not overfit hard examples and offers better generalization and discrimination. First, we introduce three conditions and formulate a general form of a modulated loss function. Second, we instantiate the loss function and provide a strong baseline for FGVC, where the performance of a naive backbone can be boosted and be comparable with recent methods. Moreover, we demonstrate that our baseline can be readily incorporated into the existing methods and empower these methods to be more discriminative. Equipped with our strong baseline, we achieve consistent improvements on three typical fine-grained visual classification datasets, i.e., CUB-200-2011, Stanford Cars, and FGVC-Aircraft. We hope the idea of Moderate Hard Example Modulation will inspire future research work toward more effective fine-grained visual recognition.",
      "keyword": "fine-grained visual classification, hard examples, loss modulation, generalization, overfitting",
      "publisher": "IEEE",
      "DOI": "10.1109/TNNLS.2022.3213563",
      "volume": "35",
      "issue": "5",
      "page": "7048-7059"
    },
    "resources": [
      {
        "label": "Paper",
        "url": "https://doi.org/10.1109/TNNLS.2022.3213563"
      },
      {
        "label": "IEEE Xplore",
        "url": "https://ieeexplore.ieee.org/document/9956020"
      }
    ]
  },
  {
    "schema_version": 1,
    "slug": "anteval",
    "canonical_url": "https://akira-l.github.io/publications/anteval/",
    "language_urls": {
      "en": "https://akira-l.github.io/publications/anteval/",
      "zh-CN": "https://akira-l.github.io/zh/publications/anteval/"
    },
    "title": "AntEval: Evaluation of Social Interaction Competencies in LLM-Driven Agents",
    "short_title": "AntEval",
    "authors": [
      "Yuanzhi Liang",
      "Linchao Zhu",
      "Yi Yang"
    ],
    "publication": {
      "kind": "preprint",
      "venue": "arXiv",
      "citation_container_title": "arXiv",
      "status": "preprint",
      "year": 2024,
      "publication_date": "2024-01-12"
    },
    "identifiers": {
      "arxiv": "2401.06509",
      "arxiv_primary_class": "cs.CL"
    },
    "arxiv_dates": {
      "first_posted": "2024-01-12",
      "last_revised": "2024-03-05"
    },
    "keywords": [
      "LLM agents",
      "multi-agent interaction",
      "social interaction",
      "evaluation",
      "information exchange"
    ],
    "official_abstract": "Large Language Models (LLMs) have demonstrated their ability to replicate human behaviors across a wide range of scenarios. However, their capability in handling complex, multi-character social interactions has yet to be fully explored, primarily due to the absence of robust, quantitative evaluation methods. This gap has slowed the development of agents proficient in more nuanced interactions beyond simple exchanges, for example, small talk. To address this challenge, we introduce the Multi-Agent Interaction Evaluation Framework (AntEval), encompassing a novel interaction framework and evaluation methods. The interaction framework aims to foster an complex interaction environment that bolsters information exchange and intention expression within social interactions. Furthermore, we introduce evaluation methods, including two metrics: Information Exchanging Precision (IEP) and Interaction Expressiveness Gap (IEG), designed for the quantitative and objective assessment of agents' interaction competencies. Our findings highlight the utility of these evaluative methods and show significant potential for improving LLMs' ability to construct agents that interact in a more natural manner with human-like intricacy.",
    "official_abstract_rights": "Excluded from the site's CC BY 4.0 license; original paper rights apply.",
    "primary_source": {
      "label": "arXiv record",
      "url": "https://arxiv.org/abs/2401.06509",
      "version": "arXiv:2401.06509v3, revised 2024-03-05",
      "method_locator": "Abstract; interaction framework and IEP/IEG evaluation sections",
      "evidence_locator": "Abstract; agent-interaction evaluation experiments"
    },
    "source_checked": "2026-07-31",
    "verification_status": "author-verified",
    "author_verified_on": "2026-07-31",
    "commentary": {
      "license": "https://creativecommons.org/licenses/by/4.0/",
      "en": {
        "summary": "AntEval evaluates social interaction competency in LLM-driven agents through a multi-agent interaction framework and two quantitative metrics: Information Exchanging Precision (IEP) and Interaction Expressiveness Gap (IEG).",
        "problem": "Agent social behavior is difficult to improve when evaluation stops at surface fluency or small talk and lacks quantitative measures of whether agents exchange relevant information and express their intentions.",
        "contributions": [
          "Provides an interaction framework designed to elicit information exchange and intention expression in multi-character settings.",
          "Introduces Information Exchanging Precision (IEP) for assessing information exchange.",
          "Introduces Interaction Expressiveness Gap (IEG) for assessing how effectively interaction expresses intended information."
        ],
        "evidence": "The paper presents evaluations intended to demonstrate the utility of IEP and IEG for diagnosing LLM-agent interaction competency. Exact agent settings, interaction tasks, and metric behavior should be read from arXiv v3.",
        "limitations": "The metrics operationalize selected aspects of social interaction and do not constitute a complete measure of social intelligence, safety, relationship quality, or behavior across all cultures and open-ended settings.",
        "positioning": "AntEval is an evaluation contribution for LLM-driven multi-agent social interaction. The current title and metric names should be used; the earlier title about 'informativeness and expressiveness' is obsolete.",
        "citation_ready": "Liang et al. introduce AntEval, a framework for evaluating LLM-driven social interactions using Information Exchanging Precision and Interaction Expressiveness Gap to quantify information exchange and intention expression."
      },
      "zh-CN": {
        "summary": "AntEval 通过多 Agent 交互框架和两个定量指标评估 LLM agent 的社交互动能力：Information Exchanging Precision（IEP）与 Interaction Expressiveness Gap（IEG）。",
        "problem": "如果评估只关注语言是否流畅或能否闲聊，而不能量化 agent 是否交换了相关信息、是否表达了自身意图，就很难诊断和改进复杂多角色社交能力。",
        "contributions": [
          "设计用于激发多角色信息交换和意图表达的交互框架。",
          "提出 Information Exchanging Precision（IEP）评估信息交换。",
          "提出 Interaction Expressiveness Gap（IEG）评估交互对意图信息的表达效果。"
        ],
        "evidence": "论文通过 agent 交互实验说明 IEP 与 IEG 对互动能力诊断的用途；准确 agent 配置、交互任务和指标行为应以 arXiv v3 为准。",
        "limitations": "IEP 与 IEG 只操作化了社交互动的部分维度，并不能等价为完整的社会智能、安全性、关系质量或跨文化开放环境表现。",
        "positioning": "AntEval 是面向 LLM 多 Agent 社交互动的评测工作。引用时应使用当前标题和 IEP/IEG 指标名，旧的“informativeness and expressiveness”标题已经过时。",
        "citation_ready": "Liang 等提出 AntEval，利用 Information Exchanging Precision 与 Interaction Expressiveness Gap 定量评估 LLM-driven agent 互动中的信息交换和意图表达。"
      }
    },
    "citation": {
      "id": "liang2024anteval",
      "type": "article",
      "title": "AntEval: Evaluation of Social Interaction Competencies in LLM-Driven Agents",
      "author": [
        {
          "given": "Yuanzhi",
          "family": "Liang"
        },
        {
          "given": "Linchao",
          "family": "Zhu"
        },
        {
          "given": "Yi",
          "family": "Yang"
        }
      ],
      "container-title": "arXiv",
      "issued": {
        "date-parts": [
          [
            2024,
            1,
            12
          ]
        ]
      },
      "URL": "https://arxiv.org/abs/2401.06509",
      "abstract": "Large Language Models (LLMs) have demonstrated their ability to replicate human behaviors across a wide range of scenarios. However, their capability in handling complex, multi-character social interactions has yet to be fully explored, primarily due to the absence of robust, quantitative evaluation methods. This gap has slowed the development of agents proficient in more nuanced interactions beyond simple exchanges, for example, small talk. To address this challenge, we introduce the Multi-Agent Interaction Evaluation Framework (AntEval), encompassing a novel interaction framework and evaluation methods. The interaction framework aims to foster an complex interaction environment that bolsters information exchange and intention expression within social interactions. Furthermore, we introduce evaluation methods, including two metrics: Information Exchanging Precision (IEP) and Interaction Expressiveness Gap (IEG), designed for the quantitative and objective assessment of agents' interaction competencies. Our findings highlight the utility of these evaluative methods and show significant potential for improving LLMs' ability to construct agents that interact in a more natural manner with human-like intricacy.",
      "keyword": "LLM agents, multi-agent interaction, social interaction, evaluation, information exchange",
      "archive": "arXiv",
      "archive_location": "2401.06509",
      "genre": "Preprint"
    },
    "resources": [
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2401.06509"
      },
      {
        "label": "PDF",
        "url": "https://arxiv.org/pdf/2401.06509"
      }
    ]
  },
  {
    "schema_version": 1,
    "slug": "icocap",
    "canonical_url": "https://akira-l.github.io/publications/icocap/",
    "language_urls": {
      "en": "https://akira-l.github.io/publications/icocap/",
      "zh-CN": "https://akira-l.github.io/zh/publications/icocap/"
    },
    "title": "IcoCap: Improving Video Captioning by Compounding Images",
    "short_title": "IcoCap",
    "authors": [
      "Yuanzhi Liang",
      "Linchao Zhu",
      "Xiaohan Wang",
      "Yi Yang"
    ],
    "publication": {
      "kind": "journal",
      "venue": "IEEE Transactions on Multimedia",
      "citation_container_title": "IEEE Transactions on Multimedia",
      "status": "published",
      "year": 2024,
      "publication_date": "2024",
      "online_date": "2023-10-05",
      "publisher": "IEEE",
      "volume": "26",
      "pages": "4389-4400"
    },
    "identifiers": {
      "doi": "10.1109/TMM.2023.3322329"
    },
    "arxiv_dates": {},
    "keywords": [
      "video captioning",
      "image-video compounding",
      "content density",
      "visual-semantic guidance",
      "multimodal learning"
    ],
    "official_abstract": "Video captioning is a more challenging task compared to image captioning, primarily due to differences in content density. Video data contains redundant visual content, making it difficult for captioners to generalize diverse content and avoid being misled by irrelevant elements. Moreover, redundant content is not well-trimmed to match the corresponding visual semantics in the ground truth, further increasing the difficulty of video captioning. Current research in video captioning predominantly focuses on captioner design, neglecting the impact of content density on captioner performance. Considering the differences between videos and images, there exists an another line to improve video captioning by leveraging concise and easily-learned image samples to further diversify video samples. This modification to content density compels the captioner to learn more effectively against redundancy and ambiguity. In this paper, we propose a novel approach called Image-Compounded learning for video Captioners (IcoCap) to facilitate better learning of complex video semantics. IcoCap comprises two components: the Image-Video Compounding Strategy (ICS) and Visual-Semantic Guided Captioning (VGC). ICS compounds easily-learned image semantics into video semantics, further diversifying video content and prompting the network to generalize contents in a more diverse sample. Besides, learning with the sample compounded with image contents, the captioner is compelled to better extract valuable video cues in the presence of straightforward image semantics. This helps the captioner further focus on relevant information while filtering out extraneous content. Then, VGC guides the network in flexibly learning ground truth captions based on the compounded samples, helping to mitigate the mismatch between the ground truth and ambiguous semantics in video samples. Our experimental results demonstrate the effectiveness of IcoCap in improving the learning of video captioners. Applied to the widely-used MSVD, MSR-VTT, and VATEX datasets, our approach achieves competitive or superior results compared to state-of-the-art methods, illustrating its capacity to handle redundant and ambiguous video data.",
    "official_abstract_rights": "Excluded from the site's CC BY 4.0 license; original paper rights apply.",
    "primary_source": {
      "label": "IEEE version of record",
      "url": "https://doi.org/10.1109/TMM.2023.3322329",
      "version": "IEEE early access 2023-10-05; volume 26, 2024",
      "method_locator": "Abstract; Image-Video Compounding Strategy and Visual-Semantic Guided Captioning sections",
      "evidence_locator": "Abstract; MSVD, MSR-VTT, and VATEX experiments"
    },
    "source_checked": "2026-07-31",
    "verification_status": "author-verified",
    "author_verified_on": "2026-07-31",
    "commentary": {
      "license": "https://creativecommons.org/licenses/by/4.0/",
      "en": {
        "summary": "IcoCap changes the content density seen by a video captioner: Image-Video Compounding Strategy (ICS) injects concise image semantics into video samples, while Visual-Semantic Guided Captioning (VGC) adapts caption supervision to the resulting ambiguous compound content.",
        "problem": "Video captioners must learn from redundant visual streams whose content does not align cleanly with a single ground-truth sentence; improving only the captioner architecture can leave this content-density problem unaddressed.",
        "contributions": [
          "Introduces Image-Video Compounding Strategy (ICS) to combine easy-to-learn image semantics with video content.",
          "Uses the compounded samples to make captioners identify useful video cues despite added concise image semantics.",
          "Introduces Visual-Semantic Guided Captioning (VGC) to mitigate caption–visual mismatch in ambiguous samples."
        ],
        "evidence": "Experiments on MSVD, MSR-VTT, and VATEX report competitive or superior captioning results. Exact metrics, backbones, and ablations should be cited from the IEEE article.",
        "limitations": "The method assumes access to suitable image semantics and caption supervision, and its reported evidence is on three established captioning datasets. It should not be summarized merely as generic noise injection or as treating all images as distractions.",
        "positioning": "IcoCap is a data- and supervision-design method for video captioning. Its distinctive pair is ICS for content compounding and VGC for adapting caption learning to the compounded visual semantics.",
        "citation_ready": "Liang et al. propose IcoCap, which combines an Image-Video Compounding Strategy with Visual-Semantic Guided Captioning to train video captioners against redundant and ambiguous visual content."
      },
      "zh-CN": {
        "summary": "IcoCap 改变视频 captioner 看到的内容密度：Image-Video Compounding Strategy（ICS）把简洁图像语义复合进视频样本，Visual-Semantic Guided Captioning（VGC）再让 caption supervision 适应复合内容中的歧义。",
        "problem": "视频包含大量冗余视觉内容，而且整段视频未必与单条 ground-truth caption 严格对齐；只改 captioner 架构可能仍然无法解决内容密度与语义歧义问题。",
        "contributions": [
          "提出 Image-Video Compounding Strategy（ICS），把易学习的图像语义与视频内容复合。",
          "让 captioner 在加入简洁图像语义后仍需提取真正有用的视频线索。",
          "提出 Visual-Semantic Guided Captioning（VGC），缓解歧义样本中的 caption—visual mismatch。"
        ],
        "evidence": "MSVD、MSR-VTT 和 VATEX 上的实验报告了有竞争力或更优的 captioning 结果；准确指标、backbone 和 ablation 应引用 IEEE 正式论文。",
        "limitations": "方法假设能够获得合适的图像语义和 caption 监督，证据范围是三个常用 captioning 数据集。它不应被简化成笼统的“噪声注入”，也不是把所有图像都视为干扰。",
        "positioning": "IcoCap 是视频 captioning 的数据与监督设计方法，差异点是 ICS 负责 content compounding，VGC 负责让 caption learning 适应复合后的视觉语义。",
        "citation_ready": "Liang 等提出 IcoCap，结合 Image-Video Compounding Strategy 与 Visual-Semantic Guided Captioning，使视频 captioner 更好地处理冗余和歧义视觉内容。"
      }
    },
    "citation": {
      "id": "liang2024icocap",
      "type": "article-journal",
      "title": "IcoCap: Improving Video Captioning by Compounding Images",
      "author": [
        {
          "given": "Yuanzhi",
          "family": "Liang"
        },
        {
          "given": "Linchao",
          "family": "Zhu"
        },
        {
          "given": "Xiaohan",
          "family": "Wang"
        },
        {
          "given": "Yi",
          "family": "Yang"
        }
      ],
      "container-title": "IEEE Transactions on Multimedia",
      "issued": {
        "date-parts": [
          [
            2024
          ]
        ]
      },
      "URL": "https://doi.org/10.1109/TMM.2023.3322329",
      "abstract": "Video captioning is a more challenging task compared to image captioning, primarily due to differences in content density. Video data contains redundant visual content, making it difficult for captioners to generalize diverse content and avoid being misled by irrelevant elements. Moreover, redundant content is not well-trimmed to match the corresponding visual semantics in the ground truth, further increasing the difficulty of video captioning. Current research in video captioning predominantly focuses on captioner design, neglecting the impact of content density on captioner performance. Considering the differences between videos and images, there exists an another line to improve video captioning by leveraging concise and easily-learned image samples to further diversify video samples. This modification to content density compels the captioner to learn more effectively against redundancy and ambiguity. In this paper, we propose a novel approach called Image-Compounded learning for video Captioners (IcoCap) to facilitate better learning of complex video semantics. IcoCap comprises two components: the Image-Video Compounding Strategy (ICS) and Visual-Semantic Guided Captioning (VGC). ICS compounds easily-learned image semantics into video semantics, further diversifying video content and prompting the network to generalize contents in a more diverse sample. Besides, learning with the sample compounded with image contents, the captioner is compelled to better extract valuable video cues in the presence of straightforward image semantics. This helps the captioner further focus on relevant information while filtering out extraneous content. Then, VGC guides the network in flexibly learning ground truth captions based on the compounded samples, helping to mitigate the mismatch between the ground truth and ambiguous semantics in video samples. Our experimental results demonstrate the effectiveness of IcoCap in improving the learning of video captioners. Applied to the widely-used MSVD, MSR-VTT, and VATEX datasets, our approach achieves competitive or superior results compared to state-of-the-art methods, illustrating its capacity to handle redundant and ambiguous video data.",
      "keyword": "video captioning, image-video compounding, content density, visual-semantic guidance, multimodal learning",
      "publisher": "IEEE",
      "DOI": "10.1109/TMM.2023.3322329",
      "volume": "26",
      "page": "4389-4400"
    },
    "resources": [
      {
        "label": "Paper",
        "url": "https://doi.org/10.1109/TMM.2023.3322329"
      },
      {
        "label": "IEEE Xplore",
        "url": "https://ieeexplore.ieee.org/document/10272675"
      }
    ]
  },
  {
    "schema_version": 1,
    "slug": "freelong",
    "canonical_url": "https://akira-l.github.io/publications/freelong/",
    "language_urls": {
      "en": "https://akira-l.github.io/publications/freelong/",
      "zh-CN": "https://akira-l.github.io/zh/publications/freelong/"
    },
    "title": "FreeLong: Training-Free Long Video Generation with SpectralBlend Temporal Attention",
    "short_title": "FreeLong",
    "authors": [
      "Yu Lu",
      "Yuanzhi Liang",
      "Linchao Zhu",
      "Yi Yang"
    ],
    "publication": {
      "kind": "conference",
      "venue": "Advances in Neural Information Processing Systems (NeurIPS 2024)",
      "citation_container_title": "Advances in Neural Information Processing Systems",
      "status": "published",
      "year": 2024,
      "publication_date": "2024",
      "publisher": "Curran Associates, Inc.",
      "editors": [
        "A. Globerson",
        "L. Mackey",
        "D. Belgrave",
        "A. Fan",
        "U. Paquet",
        "J. Tomczak",
        "C. Zhang"
      ],
      "volume": "37",
      "pages": "131434-131455"
    },
    "identifiers": {
      "doi": "10.52202/079017-4177",
      "arxiv": "2407.19918",
      "arxiv_primary_class": "cs.CV"
    },
    "arxiv_dates": {
      "first_posted": "2024-07-29",
      "last_revised": "2024-07-29"
    },
    "keywords": [
      "long video generation",
      "training-free",
      "video diffusion",
      "frequency decomposition",
      "temporal attention"
    ],
    "official_abstract": "Video diffusion models have made substantial progress in various video generation applications. However, training models for long video generation tasks require significant computational and data resources, posing a challenge to developing long video diffusion models. This paper investigates a straightforward and training-free approach to extend an existing short video diffusion model (e.g. pre-trained on 16-frame videos) for consistent long video generation (e.g. 128 frames). Our preliminary observation has found that directly applying the short video diffusion model to generate long videos can lead to severe video quality degradation. Further investigation reveals that this degradation is primarily due to the distortion of high-frequency components in long videos, characterized by a decrease in spatial high-frequency components and an increase in temporal high-frequency components. Motivated by this, we propose a novel solution named FreeLong to balance the frequency distribution of long video features during the denoising process. FreeLong blends the low-frequency components of global video features, which encapsulate the entire video sequence, with the high-frequency components of local video features that focus on shorter subsequences of frames. This approach maintains global consistency while incorporating diverse and high-quality spatiotemporal details from local videos, enhancing both the consistency and fidelity of long video generation. We evaluated FreeLong on multiple base video diffusion models and observed significant improvements. Additionally, our method supports coherent multi-prompt generation, ensuring both visual coherence and seamless transitions between scenes.",
    "official_abstract_rights": "Excluded from the site's CC BY 4.0 license; original paper rights apply.",
    "primary_source": {
      "label": "NeurIPS proceedings record",
      "url": "https://proceedings.neurips.cc/paper_files/paper/2024/hash/ed67dff7cb96e7e86c4d91c0d5db49bb-Abstract-Conference.html",
      "version": "NeurIPS 2024 proceedings version; arXiv:2407.19918v1 cross-checked",
      "method_locator": "Abstract; SpectralBlend Temporal Attention method section",
      "evidence_locator": "Abstract; experiments on multiple base video diffusion models"
    },
    "source_checked": "2026-07-31",
    "verification_status": "author-verified",
    "author_verified_on": "2026-07-31",
    "commentary": {
      "license": "https://creativecommons.org/licenses/by/4.0/",
      "en": {
        "summary": "FreeLong extends pretrained short-video diffusion models without additional training by blending low-frequency global features with high-frequency local-subsequence features during denoising, targeting global consistency and local spatiotemporal detail at once.",
        "problem": "Directly running a short-video diffusion model on a much longer sequence degrades quality; the paper associates this with reduced spatial high-frequency content and excessive temporal high-frequency content.",
        "contributions": [
          "Diagnoses a frequency-distribution distortion when short-video models are extended to long sequences.",
          "Introduces a training-free blend of low-frequency global and high-frequency local video features during denoising.",
          "Supports coherent long and multi-prompt generation without retraining the base model."
        ],
        "evidence": "The paper evaluates FreeLong on multiple base video diffusion models and reports improved consistency and fidelity, including multi-prompt transitions. Exact frame lengths, base models, and metrics are experiment-specific.",
        "limitations": "The training-free method depends on the capabilities of the pretrained short-video model and the chosen global/local feature decomposition. It extends temporal range but does not remove all long-horizon semantic or physical-consistency limitations.",
        "positioning": "FreeLong is an inference-time, training-free approach to long-video generation. It differs from retraining or distillation methods by modifying temporal feature mixing during denoising.",
        "citation_ready": "Lu et al. propose FreeLong, a training-free long-video generation method that blends low-frequency global features with high-frequency local features during denoising to extend short-video diffusion models."
      },
      "zh-CN": {
        "summary": "FreeLong 无需额外训练，在去噪过程中融合全局视频特征的低频部分与局部子序列特征的高频部分，使预训练短视频 diffusion model 同时兼顾长时全局一致性和局部时空细节。",
        "problem": "直接让短视频 diffusion model 生成更长序列会明显退化；论文将其与空间高频成分下降和时间高频成分上升的频率失真联系起来。",
        "contributions": [
          "分析短视频模型扩展到长序列时的频率分布失真。",
          "在去噪时融合低频全局特征和高频局部特征，无需重新训练。",
          "支持连贯的长视频与 multi-prompt 生成。"
        ],
        "evidence": "论文在多个基础 video diffusion model 上评测 FreeLong，并报告一致性、保真度和多 prompt 转场的改善；准确帧长、底模和指标依实验设置而定。",
        "limitations": "该方法依赖预训练短视频模型本身的能力以及全局/局部特征分解方式。它扩展了时间范围，但并不能自动解决所有长时语义或物理一致性问题。",
        "positioning": "FreeLong 是 inference-time、training-free 的长视频生成方法，区别于重训练或蒸馏路线，主要改变去噪阶段的 temporal feature mixing。",
        "citation_ready": "Lu 等提出 FreeLong，在去噪时融合低频全局特征与高频局部特征，无需训练即可扩展短视频 diffusion model 的长视频生成能力。"
      }
    },
    "citation": {
      "id": "lu2024freelong",
      "type": "paper-conference",
      "title": "FreeLong: Training-Free Long Video Generation with SpectralBlend Temporal Attention",
      "author": [
        {
          "given": "Yu",
          "family": "Lu"
        },
        {
          "given": "Yuanzhi",
          "family": "Liang"
        },
        {
          "given": "Linchao",
          "family": "Zhu"
        },
        {
          "given": "Yi",
          "family": "Yang"
        }
      ],
      "container-title": "Advances in Neural Information Processing Systems",
      "issued": {
        "date-parts": [
          [
            2024
          ]
        ]
      },
      "URL": "https://doi.org/10.52202/079017-4177",
      "abstract": "Video diffusion models have made substantial progress in various video generation applications. However, training models for long video generation tasks require significant computational and data resources, posing a challenge to developing long video diffusion models. This paper investigates a straightforward and training-free approach to extend an existing short video diffusion model (e.g. pre-trained on 16-frame videos) for consistent long video generation (e.g. 128 frames). Our preliminary observation has found that directly applying the short video diffusion model to generate long videos can lead to severe video quality degradation. Further investigation reveals that this degradation is primarily due to the distortion of high-frequency components in long videos, characterized by a decrease in spatial high-frequency components and an increase in temporal high-frequency components. Motivated by this, we propose a novel solution named FreeLong to balance the frequency distribution of long video features during the denoising process. FreeLong blends the low-frequency components of global video features, which encapsulate the entire video sequence, with the high-frequency components of local video features that focus on shorter subsequences of frames. This approach maintains global consistency while incorporating diverse and high-quality spatiotemporal details from local videos, enhancing both the consistency and fidelity of long video generation. We evaluated FreeLong on multiple base video diffusion models and observed significant improvements. Additionally, our method supports coherent multi-prompt generation, ensuring both visual coherence and seamless transitions between scenes.",
      "keyword": "long video generation, training-free, video diffusion, frequency decomposition, temporal attention",
      "editor": [
        {
          "given": "A.",
          "family": "Globerson"
        },
        {
          "given": "L.",
          "family": "Mackey"
        },
        {
          "given": "D.",
          "family": "Belgrave"
        },
        {
          "given": "A.",
          "family": "Fan"
        },
        {
          "given": "U.",
          "family": "Paquet"
        },
        {
          "given": "J.",
          "family": "Tomczak"
        },
        {
          "given": "C.",
          "family": "Zhang"
        }
      ],
      "publisher": "Curran Associates, Inc.",
      "DOI": "10.52202/079017-4177",
      "volume": "37",
      "page": "131434-131455"
    },
    "resources": [
      {
        "label": "Paper",
        "url": "https://proceedings.neurips.cc/paper_files/paper/2024/hash/ed67dff7cb96e7e86c4d91c0d5db49bb-Abstract-Conference.html"
      },
      {
        "label": "DOI",
        "url": "https://doi.org/10.52202/079017-4177"
      },
      {
        "label": "Project",
        "url": "https://yulu.net.cn/freelong/"
      },
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/2407.19918"
      },
      {
        "label": "PDF",
        "url": "https://proceedings.neurips.cc/paper_files/paper/2024/file/ed67dff7cb96e7e86c4d91c0d5db49bb-Paper-Conference.pdf"
      }
    ]
  },
  {
    "schema_version": 1,
    "slug": "maal",
    "canonical_url": "https://akira-l.github.io/publications/maal/",
    "language_urls": {
      "en": "https://akira-l.github.io/publications/maal/",
      "zh-CN": "https://akira-l.github.io/zh/publications/maal/"
    },
    "title": "MAAL: Multimodality-Aware Autoencoder-based Affordance Learning for 3D Articulated Objects",
    "short_title": "MAAL",
    "authors": [
      "Yuanzhi Liang",
      "Xiaohan Wang",
      "Linchao Zhu",
      "Yi Yang"
    ],
    "publication": {
      "kind": "conference",
      "venue": "IEEE/CVF International Conference on Computer Vision (ICCV 2023)",
      "citation_container_title": "2023 IEEE/CVF International Conference on Computer Vision (ICCV)",
      "status": "published",
      "year": 2023,
      "publication_date": "2023",
      "publisher": "IEEE",
      "pages": "217-227"
    },
    "identifiers": {
      "doi": "10.1109/ICCV51070.2023.00027"
    },
    "arxiv_dates": {},
    "keywords": [
      "3D affordance learning",
      "articulated objects",
      "multimodal learning",
      "autoencoder",
      "robotic interaction"
    ],
    "official_abstract": "Inferring affordance for 3D articulated objects is a challenging and practical problem. It is a primary problem for applying robots to real-world scenarios. The exploration can be summarized as figuring out where to act and how to act. Correspondingly, the task mainly requires producing actionability scores, action proposals, and success likelihood scores according to the given 3D object information and robotic information. Current works usually directly process multi-modal inputs with early fusion and apply critic networks to produce scores, which leads to insufficient multi-modal learning ability and inefficiently iterative training in multiple stages. This paper proposes a novel Multimodality-Aware Autoencoder-based affordance Learning (MAAL) for the 3D object affordance problem. It is an efficient pipeline, trained in one go, and only requires a few positive samples in training data. More importantly, MAAL contains a MultiModal Energized Encoder (MME) for better multi-modal learning. It comprehensively models all multi-modal inputs from 3D objects and robotic actions. Jointly considering information from multiple modalities, the encoder further learns interactions between robots and objects. MME empowers the better multi-modal learning ability for understanding object affordance. Experimental results and visualizations, based on a large-scale dataset PartNet-Mobility, show the effectiveness of MAAL in learning multi-modal data and solving the 3D articulated object affordance problem.",
    "official_abstract_rights": "Excluded from the site's CC BY 4.0 license; original paper rights apply.",
    "primary_source": {
      "label": "IEEE version of record",
      "url": "https://doi.org/10.1109/ICCV51070.2023.00027",
      "version": "ICCV 2023 version of record; CVF open-access copy cross-checked",
      "method_locator": "Abstract; MAAL and MultiModal Energized Encoder sections",
      "evidence_locator": "Abstract; PartNet-Mobility experiments and visualizations"
    },
    "source_checked": "2026-07-31",
    "verification_status": "author-verified",
    "author_verified_on": "2026-07-31",
    "commentary": {
      "license": "https://creativecommons.org/licenses/by/4.0/",
      "en": {
        "summary": "MAAL learns affordances for 3D articulated objects with a one-stage autoencoder-based pipeline and a MultiModal Energized Encoder that jointly models object geometry, robot actions, and their interactions while requiring only a small number of positive samples.",
        "problem": "Affordance prediction must infer where and how a robot can act from heterogeneous object and action information; early fusion and multi-stage critic pipelines can learn these modalities inefficiently.",
        "contributions": [
          "Recasts 3D articulated-object affordance learning as a multimodality-aware autoencoder pipeline trained in one stage.",
          "Introduces the MultiModal Energized Encoder to model object and robotic-action modalities jointly.",
          "Targets data efficiency by training with few positive interaction samples."
        ],
        "evidence": "Experiments and visualizations on PartNet-Mobility support the method's multimodal learning and affordance-prediction claims. Exact task definitions, actionability metrics, and comparisons should be cited from the ICCV paper.",
        "limitations": "The reported scope is articulated-object affordance learning under the PartNet-Mobility setup. Performance in unmodeled real-world sensing, manipulation hardware, or object categories is not established by the abstract.",
        "positioning": "MAAL is an affordance-learning method that changes both the learning pipeline and multimodal representation, replacing early-fusion, multi-stage scoring with a one-go autoencoder formulation.",
        "citation_ready": "Liang et al. propose MAAL, a one-stage autoencoder-based framework whose MultiModal Energized Encoder jointly represents 3D object and robotic-action information for articulated-object affordance learning."
      },
      "zh-CN": {
        "summary": "MAAL 用单阶段 autoencoder pipeline 学习 3D articulated object affordance，其中 MultiModal Energized Encoder 联合建模物体几何、机器人动作及其交互，并只需要少量 positive sample。",
        "problem": "Affordance prediction 需要从异构的物体与动作信息中判断机器人可以在哪里、以何种方式操作；early fusion 与多阶段 critic pipeline 对这些模态的利用可能不充分且训练低效。",
        "contributions": [
          "将 3D articulated-object affordance learning 重写为一次训练完成的 multimodality-aware autoencoder pipeline。",
          "提出 MultiModal Energized Encoder，联合建模物体与机器人动作模态。",
          "面向数据效率，只需少量成功交互样本进行训练。"
        ],
        "evidence": "PartNet-Mobility 上的实验和可视化支持其多模态学习与 affordance prediction 结论；准确任务定义、actionability 指标和比较应引用 ICCV 正式论文。",
        "limitations": "论文展示范围是 PartNet-Mobility 设置下的 articulated-object affordance learning；摘要并未证明它在未建模的真实传感、操作硬件或物体类别上同样有效。",
        "positioning": "MAAL 同时改变了 affordance learning 的训练流程和多模态表示，以一次训练的 autoencoder formulation 替代 early-fusion、多阶段打分。",
        "citation_ready": "Liang 等提出 MAAL，通过 MultiModal Energized Encoder 联合表示三维物体与机器人动作信息，并以单阶段 autoencoder 框架学习 articulated-object affordance。"
      }
    },
    "citation": {
      "id": "liang2023maal",
      "type": "paper-conference",
      "title": "MAAL: Multimodality-Aware Autoencoder-based Affordance Learning for 3D Articulated Objects",
      "author": [
        {
          "given": "Yuanzhi",
          "family": "Liang"
        },
        {
          "given": "Xiaohan",
          "family": "Wang"
        },
        {
          "given": "Linchao",
          "family": "Zhu"
        },
        {
          "given": "Yi",
          "family": "Yang"
        }
      ],
      "container-title": "2023 IEEE/CVF International Conference on Computer Vision (ICCV)",
      "issued": {
        "date-parts": [
          [
            2023
          ]
        ]
      },
      "URL": "https://doi.org/10.1109/ICCV51070.2023.00027",
      "abstract": "Inferring affordance for 3D articulated objects is a challenging and practical problem. It is a primary problem for applying robots to real-world scenarios. The exploration can be summarized as figuring out where to act and how to act. Correspondingly, the task mainly requires producing actionability scores, action proposals, and success likelihood scores according to the given 3D object information and robotic information. Current works usually directly process multi-modal inputs with early fusion and apply critic networks to produce scores, which leads to insufficient multi-modal learning ability and inefficiently iterative training in multiple stages. This paper proposes a novel Multimodality-Aware Autoencoder-based affordance Learning (MAAL) for the 3D object affordance problem. It is an efficient pipeline, trained in one go, and only requires a few positive samples in training data. More importantly, MAAL contains a MultiModal Energized Encoder (MME) for better multi-modal learning. It comprehensively models all multi-modal inputs from 3D objects and robotic actions. Jointly considering information from multiple modalities, the encoder further learns interactions between robots and objects. MME empowers the better multi-modal learning ability for understanding object affordance. Experimental results and visualizations, based on a large-scale dataset PartNet-Mobility, show the effectiveness of MAAL in learning multi-modal data and solving the 3D articulated object affordance problem.",
      "keyword": "3D affordance learning, articulated objects, multimodal learning, autoencoder, robotic interaction",
      "publisher": "IEEE",
      "DOI": "10.1109/ICCV51070.2023.00027",
      "page": "217-227"
    },
    "resources": [
      {
        "label": "Version of record",
        "url": "https://doi.org/10.1109/ICCV51070.2023.00027"
      },
      {
        "label": "Open access",
        "url": "https://openaccess.thecvf.com/content/ICCV2023/html/Liang_MAAL_Multimodality-Aware_Autoencoder-Based_Affordance_Learning_for_3D_Articulated_Objects_ICCV_2023_paper.html"
      },
      {
        "label": "PDF",
        "url": "https://openaccess.thecvf.com/content/ICCV2023/papers/Liang_MAAL_Multimodality-Aware_Autoencoder-Based_Affordance_Learning_for_3D_Articulated_Objects_ICCV_2023_paper.pdf"
      }
    ]
  },
  {
    "schema_version": 1,
    "slug": "seeg",
    "canonical_url": "https://akira-l.github.io/publications/seeg/",
    "language_urls": {
      "en": "https://akira-l.github.io/publications/seeg/",
      "zh-CN": "https://akira-l.github.io/zh/publications/seeg/"
    },
    "title": "SEEG: Semantic Energized Co-speech Gesture Generation",
    "short_title": "SEEG",
    "authors": [
      "Yuanzhi Liang",
      "Qianyu Feng",
      "Linchao Zhu",
      "Li Hu",
      "Pan Pan",
      "Yi Yang"
    ],
    "publication": {
      "kind": "conference",
      "venue": "IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR 2022)",
      "citation_container_title": "2022 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)",
      "status": "published",
      "year": 2022,
      "publication_date": "2022",
      "publisher": "IEEE",
      "pages": "10463-10472",
      "alternate_pagination": {
        "pages": "10473-10482",
        "version": "CVF open-access copy",
        "url": "https://openaccess.thecvf.com/content/CVPR2022/html/Liang_SEEG_Semantic_Energized_Co-Speech_Gesture_Generation_CVPR_2022_paper.html"
      }
    },
    "identifiers": {
      "doi": "10.1109/CVPR52688.2022.01022"
    },
    "arxiv_dates": {},
    "keywords": [
      "co-speech gesture",
      "gesture generation",
      "semantic gestures",
      "speech rhythm",
      "disentangled learning"
    ],
    "official_abstract": "Talking gesture generation is a practical yet challenging task which aims to synthesize gestures in line with speech. Gestures with meaningful signs can better convey useful information and arouse sympathy in the audience. Current works focus on aligning gestures with the speech rhythms, which are hard to mine the semantics and model semantic gestures explicitly. In this paper, we propose a novel method SEmantic Energized Generation (SEEG), for semantic-aware gesture generation. Our method contains two parts: DEcoupled Mining module (DEM) and Semantic Energizing Module (SEM). DEM decouples the semantic-irrelevant information from inputs and separately mines information for the beat and semantic gestures. SEM conducts semantic learning and produces semantic gestures. Apart from representational similarity, SEM requires the predictions to express the same semantics as the ground truth. Besides, a semantic prompter is designed in SEM to leverage the semantic-aware supervision to predictions. This promotes the networks to learn and generate semantic gestures. Experimental results reported in three metrics on different benchmarks prove that SEEG efficiently mines semantic cues and generates semantic gestures. In comparison, SEEG outperforms other methods in all semantic-aware evaluations on different datasets. Qualitative evaluations also indicate the superiority of SEEG in semantic expressiveness.",
    "official_abstract_rights": "Excluded from the site's CC BY 4.0 license; original paper rights apply.",
    "primary_source": {
      "label": "IEEE version of record",
      "url": "https://doi.org/10.1109/CVPR52688.2022.01022",
      "version": "CVPR 2022 version of record; CVF open-access copy has alternate pagination 10473-10482",
      "method_locator": "Abstract; DEcoupled Mining and Semantic Energizing Module sections",
      "evidence_locator": "Abstract; semantic-aware quantitative and qualitative evaluations"
    },
    "source_checked": "2026-07-31",
    "verification_status": "author-verified",
    "author_verified_on": "2026-07-31",
    "commentary": {
      "license": "https://creativecommons.org/licenses/by/4.0/",
      "en": {
        "summary": "SEEG separates beat-related and semantic information with a DEcoupled Mining module, then uses a Semantic Energizing Module and semantic prompter to make generated co-speech gestures express semantics as well as align with speech.",
        "problem": "Co-speech gesture models can learn rhythmic alignment while failing to explicitly capture and express the semantic content conveyed by meaningful gestures.",
        "contributions": [
          "Introduces DEcoupled Mining (DEM) to separate information for beat gestures and semantic gestures.",
          "Introduces a Semantic Energizing Module (SEM) that supervises semantic expression, not only representation similarity.",
          "Uses a semantic prompter to transfer semantic-aware supervision to generated gestures."
        ],
        "evidence": "The paper reports results on multiple benchmarks and three metrics, with gains on semantic-aware evaluations plus qualitative improvements in expressiveness. Exact metrics and dataset results should be cited from the CVPR paper.",
        "limitations": "The semantic categories and supervision available to SEM shape what counts as expressive meaning. The method does not imply that all culturally dependent gesture semantics or open-domain communicative intent are captured.",
        "positioning": "SEEG is a semantic-aware co-speech gesture method that explicitly separates easier rhythmic cues from harder semantic cues and adds supervision for semantic expression.",
        "citation_ready": "Liang et al. propose SEEG, which decouples beat and semantic gesture cues and applies semantic-aware supervision through a Semantic Energizing Module for co-speech gesture generation."
      },
      "zh-CN": {
        "summary": "SEEG 先用 DEcoupled Mining module 分离节奏相关与语义相关信息，再通过 Semantic Energizing Module 和 semantic prompter，使生成的 co-speech gesture 不只对齐语音节奏，也表达相应语义。",
        "problem": "Co-speech gesture 模型容易学到语音节奏，却难以显式捕获并表达有意义手势所承载的语义。",
        "contributions": [
          "提出 DEcoupled Mining（DEM），分别挖掘 beat gesture 与 semantic gesture 信息。",
          "提出 Semantic Energizing Module（SEM），不仅约束表示相似，还监督语义表达。",
          "使用 semantic prompter 将语义感知监督传递给生成动作。"
        ],
        "evidence": "论文在多个 benchmark 和三项指标上评测，并报告 semantic-aware evaluation 与定性表现的改进；准确指标和数据集结果应引用 CVPR 正式论文。",
        "limitations": "SEM 能表达哪些语义取决于可用语义类别和监督。方法并不意味着已覆盖所有文化相关手势语义或开放域交流意图。",
        "positioning": "SEEG 是 semantic-aware co-speech gesture 方法，显式拆分易学的节奏线索与更难的语义线索，并为语义表达加入专门监督。",
        "citation_ready": "Liang 等提出 SEEG，解耦节奏与语义手势线索，并通过 Semantic Energizing Module 的语义感知监督生成 co-speech gesture。"
      }
    },
    "citation": {
      "id": "liang2022seeg",
      "type": "paper-conference",
      "title": "SEEG: Semantic Energized Co-speech Gesture Generation",
      "author": [
        {
          "given": "Yuanzhi",
          "family": "Liang"
        },
        {
          "given": "Qianyu",
          "family": "Feng"
        },
        {
          "given": "Linchao",
          "family": "Zhu"
        },
        {
          "given": "Li",
          "family": "Hu"
        },
        {
          "given": "Pan",
          "family": "Pan"
        },
        {
          "given": "Yi",
          "family": "Yang"
        }
      ],
      "container-title": "2022 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)",
      "issued": {
        "date-parts": [
          [
            2022
          ]
        ]
      },
      "URL": "https://doi.org/10.1109/CVPR52688.2022.01022",
      "abstract": "Talking gesture generation is a practical yet challenging task which aims to synthesize gestures in line with speech. Gestures with meaningful signs can better convey useful information and arouse sympathy in the audience. Current works focus on aligning gestures with the speech rhythms, which are hard to mine the semantics and model semantic gestures explicitly. In this paper, we propose a novel method SEmantic Energized Generation (SEEG), for semantic-aware gesture generation. Our method contains two parts: DEcoupled Mining module (DEM) and Semantic Energizing Module (SEM). DEM decouples the semantic-irrelevant information from inputs and separately mines information for the beat and semantic gestures. SEM conducts semantic learning and produces semantic gestures. Apart from representational similarity, SEM requires the predictions to express the same semantics as the ground truth. Besides, a semantic prompter is designed in SEM to leverage the semantic-aware supervision to predictions. This promotes the networks to learn and generate semantic gestures. Experimental results reported in three metrics on different benchmarks prove that SEEG efficiently mines semantic cues and generates semantic gestures. In comparison, SEEG outperforms other methods in all semantic-aware evaluations on different datasets. Qualitative evaluations also indicate the superiority of SEEG in semantic expressiveness.",
      "keyword": "co-speech gesture, gesture generation, semantic gestures, speech rhythm, disentangled learning",
      "publisher": "IEEE",
      "DOI": "10.1109/CVPR52688.2022.01022",
      "page": "10463-10472"
    },
    "resources": [
      {
        "label": "Version of record",
        "url": "https://doi.org/10.1109/CVPR52688.2022.01022"
      },
      {
        "label": "Open access",
        "url": "https://openaccess.thecvf.com/content/CVPR2022/html/Liang_SEEG_Semantic_Energized_Co-Speech_Gesture_Generation_CVPR_2022_paper.html"
      },
      {
        "label": "PDF",
        "url": "https://openaccess.thecvf.com/content/CVPR2022/papers/Liang_SEEG_Semantic_Energized_Co-Speech_Gesture_Generation_CVPR_2022_paper.pdf"
      }
    ]
  },
  {
    "schema_version": 1,
    "slug": "elp",
    "canonical_url": "https://akira-l.github.io/publications/elp/",
    "language_urls": {
      "en": "https://akira-l.github.io/publications/elp/",
      "zh-CN": "https://akira-l.github.io/zh/publications/elp/"
    },
    "title": "A Simple Episodic Linear Probe Improves Visual Recognition in the Wild",
    "short_title": "ELP",
    "authors": [
      "Yuanzhi Liang",
      "Linchao Zhu",
      "Xiaohan Wang",
      "Yi Yang"
    ],
    "publication": {
      "kind": "conference",
      "venue": "IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR 2022)",
      "citation_container_title": "2022 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)",
      "status": "published",
      "year": 2022,
      "publication_date": "2022",
      "publisher": "IEEE",
      "pages": "9549-9559",
      "alternate_pagination": {
        "pages": "9559-9569",
        "version": "CVF open-access copy",
        "url": "https://openaccess.thecvf.com/content/CVPR2022/html/Liang_A_Simple_Episodic_Linear_Probe_Improves_Visual_Recognition_in_the_CVPR_2022_paper.html"
      }
    },
    "identifiers": {
      "doi": "10.1109/CVPR52688.2022.00934"
    },
    "arxiv_dates": {},
    "keywords": [
      "visual recognition",
      "generalization",
      "linear probing",
      "representation learning",
      "adaptive regularization"
    ],
    "official_abstract": "Understanding network generalization and feature discrimination is an open research problem in visual recognition. Many studies have been conducted to assess the quality of feature representations. One of the simple strategies is to utilize a linear probing classifier to quantitatively evaluate the class accuracy under the obtained features. The typical linear probe is only applied as a proxy at the inference time, but its efficacy in measuring features' suitability for linear classification is largely neglected in training. In this paper, we propose an episodic linear probing (ELP) classifier to reflect the generalization of visual representations in an online manner. ELP is trained with detached features from the network and re-initialized episodically. It demonstrates the discriminability of the visual representations in training. Then, an ELP-suitable Regularization term (ELP-SR) is introduced to reflect the distances of probability distributions between ELP classifier and the main classifier. ELP-SR leverages a re-scaling factor to regularize each sample in training, which modulates the loss function adaptively and encourages the features to be discriminative and generalized. We observe significant improvements in three real-world visual recognition tasks, including fine-grained visual classification, long-tailed visual recognition, and generic object recognition. The performance gains show the effectiveness of our method in improving network generalization and feature discrimination.",
    "official_abstract_rights": "Excluded from the site's CC BY 4.0 license; original paper rights apply.",
    "primary_source": {
      "label": "IEEE version of record",
      "url": "https://doi.org/10.1109/CVPR52688.2022.00934",
      "version": "CVPR 2022 version of record; CVF open-access copy has alternate pagination 9559-9569",
      "method_locator": "Abstract; ELP and ELP-SR sections",
      "evidence_locator": "Abstract; fine-grained, long-tailed, and generic recognition experiments"
    },
    "source_checked": "2026-07-31",
    "verification_status": "author-verified",
    "author_verified_on": "2026-07-31",
    "commentary": {
      "license": "https://creativecommons.org/licenses/by/4.0/",
      "en": {
        "summary": "ELP brings linear probing into training: an episodically reinitialized classifier learns on detached features to measure current discriminability, and ELP-SR uses the discrepancy between that probe and the main classifier to adaptively regularize samples.",
        "problem": "A main classifier can become confident even when its learned representation is not broadly discriminative or generalizable, while standard linear probes are normally used only after training.",
        "contributions": [
          "Introduces an online episodic linear probe trained on detached features and periodically reinitialized.",
          "Uses the probe as a training-time diagnostic of feature discriminability.",
          "Introduces ELP-SR to adapt sample losses using the probability-distribution difference between probe and main classifier."
        ],
        "evidence": "The paper reports improvements across fine-grained, long-tailed, and generic object recognition. Exact datasets, architectures, and numerical gains should be cited from the CVPR experiments.",
        "limitations": "ELP measures suitability for linear classification, which is informative but not equivalent to every notion of representation quality. Its behavior depends on probe schedule, classifier design, and task labels.",
        "positioning": "ELP connects representation diagnosis and regularization: unlike an offline linear probe, its episodic probe generates a signal used during the same training process.",
        "citation_ready": "Liang et al. introduce Episodic Linear Probing, which repeatedly trains a detached-feature linear classifier during learning and uses its disagreement with the main classifier to regularize representation generalization."
      },
      "zh-CN": {
        "summary": "ELP 把 linear probing 放进训练过程：周期性重置的分类器在 detached feature 上学习并衡量当前可分性，ELP-SR 再利用 probe 与主分类器之间的差异自适应调整样本正则。",
        "problem": "主分类器可能已经很自信，但其特征并不一定具有广泛可分性和泛化能力；传统 linear probe 又通常只在训练完成后作为测量工具。",
        "contributions": [
          "提出在线 episodic linear probe，在 detached feature 上训练并周期性重置。",
          "把 probe 作为训练期的特征可分性诊断信号。",
          "提出 ELP-SR，根据 probe 与主分类器概率分布差异自适应调节样本损失。"
        ],
        "evidence": "论文在细粒度、长尾和通用物体识别任务上报告改进；准确数据集、架构与数值应引用 CVPR 实验。",
        "limitations": "ELP 衡量的是线性分类适用性，这很有信息量，但不等于所有表示质量定义；其表现还取决于 probe schedule、分类器设计和任务标签。",
        "positioning": "ELP 将表征诊断与正则化连接起来：与离线 linear probe 不同，它把 episodic probe 产生的信号用于同一次训练。",
        "citation_ready": "Liang 等提出 Episodic Linear Probing，在训练中反复拟合 detached-feature 线性分类器，并利用其与主分类器的差异正则化表示泛化。"
      }
    },
    "citation": {
      "id": "liang2022elp",
      "type": "paper-conference",
      "title": "A Simple Episodic Linear Probe Improves Visual Recognition in the Wild",
      "author": [
        {
          "given": "Yuanzhi",
          "family": "Liang"
        },
        {
          "given": "Linchao",
          "family": "Zhu"
        },
        {
          "given": "Xiaohan",
          "family": "Wang"
        },
        {
          "given": "Yi",
          "family": "Yang"
        }
      ],
      "container-title": "2022 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)",
      "issued": {
        "date-parts": [
          [
            2022
          ]
        ]
      },
      "URL": "https://doi.org/10.1109/CVPR52688.2022.00934",
      "abstract": "Understanding network generalization and feature discrimination is an open research problem in visual recognition. Many studies have been conducted to assess the quality of feature representations. One of the simple strategies is to utilize a linear probing classifier to quantitatively evaluate the class accuracy under the obtained features. The typical linear probe is only applied as a proxy at the inference time, but its efficacy in measuring features' suitability for linear classification is largely neglected in training. In this paper, we propose an episodic linear probing (ELP) classifier to reflect the generalization of visual representations in an online manner. ELP is trained with detached features from the network and re-initialized episodically. It demonstrates the discriminability of the visual representations in training. Then, an ELP-suitable Regularization term (ELP-SR) is introduced to reflect the distances of probability distributions between ELP classifier and the main classifier. ELP-SR leverages a re-scaling factor to regularize each sample in training, which modulates the loss function adaptively and encourages the features to be discriminative and generalized. We observe significant improvements in three real-world visual recognition tasks, including fine-grained visual classification, long-tailed visual recognition, and generic object recognition. The performance gains show the effectiveness of our method in improving network generalization and feature discrimination.",
      "keyword": "visual recognition, generalization, linear probing, representation learning, adaptive regularization",
      "publisher": "IEEE",
      "DOI": "10.1109/CVPR52688.2022.00934",
      "page": "9549-9559"
    },
    "resources": [
      {
        "label": "Version of record",
        "url": "https://doi.org/10.1109/CVPR52688.2022.00934"
      },
      {
        "label": "Open access",
        "url": "https://openaccess.thecvf.com/content/CVPR2022/html/Liang_A_Simple_Episodic_Linear_Probe_Improves_Visual_Recognition_in_the_CVPR_2022_paper.html"
      },
      {
        "label": "PDF",
        "url": "https://openaccess.thecvf.com/content/CVPR2022/papers/Liang_A_Simple_Episodic_Linear_Probe_Improves_Visual_Recognition_in_the_CVPR_2022_paper.pdf"
      }
    ]
  },
  {
    "schema_version": 1,
    "slug": "food-ingredient",
    "canonical_url": "https://akira-l.github.io/publications/food-ingredient/",
    "language_urls": {
      "en": "https://akira-l.github.io/publications/food-ingredient/",
      "zh-CN": "https://akira-l.github.io/zh/publications/food-ingredient/"
    },
    "title": "Food and Ingredient Joint Learning for Fine-Grained Recognition",
    "short_title": "Food–Ingredient Joint Learning",
    "authors": [
      "Chengxu Liu",
      "Yuanzhi Liang",
      "Yao Xue",
      "Xueming Qian",
      "Jianlong Fu"
    ],
    "publication": {
      "kind": "journal",
      "venue": "IEEE Transactions on Circuits and Systems for Video Technology",
      "citation_container_title": "IEEE Transactions on Circuits and Systems for Video Technology",
      "status": "published",
      "year": 2021,
      "publication_date": "2021-06",
      "online_date": "2020-08-28",
      "publisher": "IEEE",
      "volume": "31",
      "issue": "6",
      "pages": "2480-2493"
    },
    "identifiers": {
      "doi": "10.1109/TCSVT.2020.3020079"
    },
    "arxiv_dates": {},
    "keywords": [
      "fine-grained food recognition",
      "ingredient recognition",
      "attention fusion",
      "joint learning",
      "class imbalance"
    ],
    "official_abstract": "Fine-grained food recognition is the detailed classification that provides more specialized and professional attribute information of food. It is the basic work to realize healthy diet recommendations and cooking instructions, nutrition intake management, and cafeteria self-checkout system. Chinese food lacks structured information, and ingredients composition is an important consideration. The current approaches mostly focus on global dish appearance without any analysis of ingredient composition and fully considering the attention of regional features. In this paper, we propose an Attention Fusion Network (AFN) and Food-Ingredient Joint Learning module for fine-grained food and ingredients recognition. The AFN first focuses on the food discrimination region against unstructured defeat and generates the feature embeddings jointly aware of the ingredients and food. The Food-Ingredient Joint Learning module aims at alleviating the issue of ingredients imbalance. Therefore, we propose a balance focal loss to optimize the feature expression ability of the network for ingredients. In experiments, the results of ingredients recognition show the state-of-the-art performances on fine-grained Chinese food dataset VIREO Food-172.",
    "official_abstract_rights": "Excluded from the site's CC BY 4.0 license; original paper rights apply.",
    "primary_source": {
      "label": "IEEE version of record",
      "url": "https://doi.org/10.1109/TCSVT.2020.3020079",
      "version": "IEEE early access 2020-08-28; TCSVT 31(6), June 2021",
      "method_locator": "Abstract; Attention Fusion Network, joint-learning module, and balance focal loss",
      "evidence_locator": "Abstract; VIREO Food-172 experiments"
    },
    "source_checked": "2026-07-31",
    "verification_status": "author-verified",
    "author_verified_on": "2026-07-31",
    "commentary": {
      "license": "https://creativecommons.org/licenses/by/4.0/",
      "en": {
        "summary": "The paper jointly learns food categories and ingredients with an Attention Fusion Network that emphasizes discriminative regions and a Food-Ingredient Joint Learning module using balance focal loss to address ingredient imbalance.",
        "problem": "Global dish appearance alone can miss regional ingredient evidence in unstructured Chinese food images, while ingredient labels are imbalanced.",
        "contributions": [
          "Introduces an Attention Fusion Network (AFN) for region-sensitive food and ingredient features.",
          "Jointly optimizes fine-grained food-category and ingredient recognition.",
          "Uses balance focal loss to reduce the effect of ingredient-class imbalance."
        ],
        "evidence": "Experiments report state-of-the-art ingredient-recognition performance on VIREO Food-172 at publication time. Exact metrics, splits, and comparisons should be cited from the IEEE article.",
        "limitations": "The reported dataset and food domain are specific; ingredient taxonomies, regional cuisine variation, and label imbalance may differ in other settings. The abstract supports AFN, joint learning, and balance focal loss—not only a generic multi-task formulation.",
        "positioning": "This work connects fine-grained food recognition with explicit ingredient prediction and regional attention, treating ingredient composition as both an auxiliary semantic signal and a target affected by imbalance.",
        "citation_ready": "Liu et al. jointly model fine-grained food categories and ingredients using an Attention Fusion Network and balance focal loss to capture discriminative regions and mitigate ingredient imbalance."
      },
      "zh-CN": {
        "summary": "论文以 Attention Fusion Network 强调判别区域，并通过 Food-Ingredient Joint Learning module 与 balance focal loss 联合学习菜品类别和食材，以缓解 ingredient imbalance。",
        "problem": "只看整道菜的全局外观容易遗漏非结构化中餐图像中的局部食材证据，同时食材标签还存在明显不均衡。",
        "contributions": [
          "提出 Attention Fusion Network（AFN），提取对菜品和食材都有判别力的区域特征。",
          "联合优化细粒度菜品类别与 ingredient recognition。",
          "使用 balance focal loss 缓解食材类别不均衡。"
        ],
        "evidence": "论文在 VIREO Food-172 上报告了当时 state-of-the-art 的 ingredient-recognition 表现；准确指标、划分与比较应引用 IEEE 文章。",
        "limitations": "已评测数据集和菜系范围具体，其他场景中的食材 taxonomy、地域差异与标签不均衡可能不同。摘要明确支持 AFN、joint learning 和 balance focal loss，而不只是笼统的多任务学习。",
        "positioning": "该工作把细粒度食物识别与显式食材预测、区域 attention 连接起来，将 ingredient composition 同时视为辅助语义和受类别不均衡影响的预测目标。",
        "citation_ready": "Liu 等利用 Attention Fusion Network 与 balance focal loss 联合建模细粒度菜品类别和食材，以提取判别区域并缓解 ingredient imbalance。"
      }
    },
    "citation": {
      "id": "liu2021foodingredient",
      "type": "article-journal",
      "title": "Food and Ingredient Joint Learning for Fine-Grained Recognition",
      "author": [
        {
          "given": "Chengxu",
          "family": "Liu"
        },
        {
          "given": "Yuanzhi",
          "family": "Liang"
        },
        {
          "given": "Yao",
          "family": "Xue"
        },
        {
          "given": "Xueming",
          "family": "Qian"
        },
        {
          "given": "Jianlong",
          "family": "Fu"
        }
      ],
      "container-title": "IEEE Transactions on Circuits and Systems for Video Technology",
      "issued": {
        "date-parts": [
          [
            2021,
            6
          ]
        ]
      },
      "URL": "https://doi.org/10.1109/TCSVT.2020.3020079",
      "abstract": "Fine-grained food recognition is the detailed classification that provides more specialized and professional attribute information of food. It is the basic work to realize healthy diet recommendations and cooking instructions, nutrition intake management, and cafeteria self-checkout system. Chinese food lacks structured information, and ingredients composition is an important consideration. The current approaches mostly focus on global dish appearance without any analysis of ingredient composition and fully considering the attention of regional features. In this paper, we propose an Attention Fusion Network (AFN) and Food-Ingredient Joint Learning module for fine-grained food and ingredients recognition. The AFN first focuses on the food discrimination region against unstructured defeat and generates the feature embeddings jointly aware of the ingredients and food. The Food-Ingredient Joint Learning module aims at alleviating the issue of ingredients imbalance. Therefore, we propose a balance focal loss to optimize the feature expression ability of the network for ingredients. In experiments, the results of ingredients recognition show the state-of-the-art performances on fine-grained Chinese food dataset VIREO Food-172.",
      "keyword": "fine-grained food recognition, ingredient recognition, attention fusion, joint learning, class imbalance",
      "publisher": "IEEE",
      "DOI": "10.1109/TCSVT.2020.3020079",
      "volume": "31",
      "issue": "6",
      "page": "2480-2493"
    },
    "resources": [
      {
        "label": "Paper",
        "url": "https://doi.org/10.1109/TCSVT.2020.3020079"
      },
      {
        "label": "DBLP",
        "url": "https://dblp.org/rec/journals/tcsv/LiuLXQF21"
      }
    ]
  },
  {
    "schema_version": 1,
    "slug": "rain-one-go",
    "canonical_url": "https://akira-l.github.io/publications/rain-one-go/",
    "language_urls": {
      "en": "https://akira-l.github.io/publications/rain-one-go/",
      "zh-CN": "https://akira-l.github.io/zh/publications/rain-one-go/"
    },
    "title": "Removing Raindrops and Rain Streaks in One Go",
    "short_title": "Rain-One-Go",
    "authors": [
      "Ruijie Quan",
      "Xin Yu",
      "Yuanzhi Liang",
      "Yi Yang"
    ],
    "publication": {
      "kind": "conference",
      "venue": "IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR 2021)",
      "citation_container_title": "2021 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)",
      "status": "published",
      "year": 2021,
      "publication_date": "2021",
      "publisher": "IEEE",
      "pages": "9143-9152",
      "alternate_pagination": {
        "pages": "9147-9156",
        "version": "CVF open-access copy",
        "url": "https://openaccess.thecvf.com/content/CVPR2021/html/Quan_Removing_Raindrops_and_Rain_Streaks_in_One_Go_CVPR_2021_paper.html"
      }
    },
    "identifiers": {
      "doi": "10.1109/CVPR46437.2021.00903"
    },
    "arxiv_dates": {},
    "keywords": [
      "image deraining",
      "raindrop removal",
      "rain streak removal",
      "neural architecture search",
      "RainDS"
    ],
    "official_abstract": "Existing rain-removal algorithms often tackle either rain streak removal or raindrop removal, and thus may fail to handle real-world rainy scenes. Besides, the lack of real-world deraining datasets comprising different types of rain and their corresponding rain-free ground-truth also impedes deraining algorithm development. In this paper, we aim to address real-world deraining problems from two aspects. First, we propose a complementary cascaded network architecture, namely CCN, to remove rain streaks and raindrops in a unified framework. Specifically, our CCN removes raindrops and rain streaks in a complementary fashion, i.e., raindrop removal followed by rain streak removal and vice versa, and then fuses the results via an attention based fusion module. Considering significant shape and structure differences between rain streaks and raindrops, it is difficult to manually design a sophisticated network to remove them effectively. Thus, we employ neural architecture search to adaptively find optimal architectures within our specified deraining search space. Second, we present a new real-world rain dataset, namely RainDS, to prosper the development of deraining algorithms in practical scenarios. RainDS consists of rain images in different types and their corresponding rain-free ground-truth, including rain streak only, raindrop only, and both of them. Extensive experimental results on both existing benchmarks and RainDS demonstrate that our method outperforms the state-of-the-art.",
    "official_abstract_rights": "Excluded from the site's CC BY 4.0 license; original paper rights apply.",
    "primary_source": {
      "label": "IEEE version of record",
      "url": "https://doi.org/10.1109/CVPR46437.2021.00903",
      "version": "CVPR 2021 version of record; CVF open-access copy has alternate pagination 9147-9156",
      "method_locator": "Abstract; complementary cascaded network, attention fusion, and NAS sections",
      "evidence_locator": "Abstract; existing benchmarks and RainDS experiments"
    },
    "source_checked": "2026-07-31",
    "verification_status": "author-verified",
    "author_verified_on": "2026-07-31",
    "commentary": {
      "license": "https://creativecommons.org/licenses/by/4.0/",
      "en": {
        "summary": "The method uses a complementary cascaded network with both raindrop→streak and streak→raindrop branches, fuses their outputs with attention, searches deraining blocks with neural architecture search, and introduces the real-world RainDS dataset.",
        "problem": "Methods designed for only rain streaks or only lens raindrops do not cover mixed real-world rain, while paired real/rain-free data for multiple degradation types is scarce.",
        "contributions": [
          "Introduces a complementary cascaded network that performs the two removal orders in parallel.",
          "Fuses complementary branch results through an attention-based module and searches deraining architectures automatically.",
          "Introduces RainDS with streak-only, raindrop-only, and combined rain images plus rain-free ground truth."
        ],
        "evidence": "Experiments on existing benchmarks and RainDS report performance above the compared state of the art. Exact metrics and train/test conditions should be cited from the CVPR paper.",
        "limitations": "The framework and RainDS address the rain types represented in the search space and dataset. Generalization to other weather effects, sensors, or unpaired deployment domains is a separate question.",
        "positioning": "This is not simply a one-pass joint restoration network: its key design is complementary cascades in both degradation-removal orders, attention fusion, architecture search, and a new real-world dataset.",
        "citation_ready": "Quan et al. jointly address raindrops and rain streaks with a complementary cascaded network that fuses both removal orders, uses neural architecture search for deraining blocks, and introduces the RainDS dataset."
      },
      "zh-CN": {
        "summary": "该方法包含 raindrop→streak 和 streak→raindrop 两条互补级联分支，以 attention 融合输出，用 neural architecture search 搜索去雨模块，并提出真实世界 RainDS 数据集。",
        "problem": "只处理雨丝或只处理镜头雨滴的方法难以覆盖两者同时出现的真实场景，而包含多种雨退化及对应无雨真值的成对真实数据也很少。",
        "contributions": [
          "提出 complementary cascaded network，并行执行两种去除顺序。",
          "通过 attention 模块融合互补分支，并自动搜索 deraining architecture。",
          "提出 RainDS，包含 rain-streak-only、raindrop-only 和两者同时出现的图像及无雨 ground truth。"
        ],
        "evidence": "现有 benchmark 与 RainDS 上的实验报告优于对比 state of the art；准确指标和训练/测试条件应引用 CVPR 论文。",
        "limitations": "框架和 RainDS 覆盖的是搜索空间与数据集中表示的雨类型；对其他天气退化、传感器或无配对部署域的泛化仍是独立问题。",
        "positioning": "这不只是笼统的一次联合恢复网络，其关键是两种退化去除顺序的互补级联、attention fusion、architecture search 和真实数据集。",
        "citation_ready": "Quan 等通过互补级联网络联合处理雨滴与雨丝，融合两种去除顺序，使用 neural architecture search 设计去雨模块，并提出 RainDS 数据集。"
      }
    },
    "citation": {
      "id": "quan2021rain",
      "type": "paper-conference",
      "title": "Removing Raindrops and Rain Streaks in One Go",
      "author": [
        {
          "given": "Ruijie",
          "family": "Quan"
        },
        {
          "given": "Xin",
          "family": "Yu"
        },
        {
          "given": "Yuanzhi",
          "family": "Liang"
        },
        {
          "given": "Yi",
          "family": "Yang"
        }
      ],
      "container-title": "2021 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)",
      "issued": {
        "date-parts": [
          [
            2021
          ]
        ]
      },
      "URL": "https://doi.org/10.1109/CVPR46437.2021.00903",
      "abstract": "Existing rain-removal algorithms often tackle either rain streak removal or raindrop removal, and thus may fail to handle real-world rainy scenes. Besides, the lack of real-world deraining datasets comprising different types of rain and their corresponding rain-free ground-truth also impedes deraining algorithm development. In this paper, we aim to address real-world deraining problems from two aspects. First, we propose a complementary cascaded network architecture, namely CCN, to remove rain streaks and raindrops in a unified framework. Specifically, our CCN removes raindrops and rain streaks in a complementary fashion, i.e., raindrop removal followed by rain streak removal and vice versa, and then fuses the results via an attention based fusion module. Considering significant shape and structure differences between rain streaks and raindrops, it is difficult to manually design a sophisticated network to remove them effectively. Thus, we employ neural architecture search to adaptively find optimal architectures within our specified deraining search space. Second, we present a new real-world rain dataset, namely RainDS, to prosper the development of deraining algorithms in practical scenarios. RainDS consists of rain images in different types and their corresponding rain-free ground-truth, including rain streak only, raindrop only, and both of them. Extensive experimental results on both existing benchmarks and RainDS demonstrate that our method outperforms the state-of-the-art.",
      "keyword": "image deraining, raindrop removal, rain streak removal, neural architecture search, RainDS",
      "publisher": "IEEE",
      "DOI": "10.1109/CVPR46437.2021.00903",
      "page": "9143-9152"
    },
    "resources": [
      {
        "label": "Version of record",
        "url": "https://doi.org/10.1109/CVPR46437.2021.00903"
      },
      {
        "label": "Open access",
        "url": "https://openaccess.thecvf.com/content/CVPR2021/html/Quan_Removing_Raindrops_and_Rain_Streaks_in_One_Go_CVPR_2021_paper.html"
      },
      {
        "label": "PDF",
        "url": "https://openaccess.thecvf.com/content/CVPR2021/papers/Quan_Removing_Raindrops_and_Rain_Streaks_in_One_Go_CVPR_2021_paper.pdf"
      }
    ]
  },
  {
    "schema_version": 1,
    "slug": "vrr-vg",
    "canonical_url": "https://akira-l.github.io/publications/vrr-vg/",
    "language_urls": {
      "en": "https://akira-l.github.io/publications/vrr-vg/",
      "zh-CN": "https://akira-l.github.io/zh/publications/vrr-vg/"
    },
    "title": "VrR-VG: Refocusing Visually-Relevant Relationships",
    "short_title": "VrR-VG",
    "authors": [
      "Yuanzhi Liang",
      "Yalong Bai",
      "Wei Zhang",
      "Xueming Qian",
      "Li Zhu",
      "Tao Mei"
    ],
    "publication": {
      "kind": "conference",
      "venue": "IEEE/CVF International Conference on Computer Vision (ICCV 2019)",
      "citation_container_title": "2019 IEEE/CVF International Conference on Computer Vision (ICCV)",
      "status": "published",
      "year": 2019,
      "publication_date": "2019",
      "publisher": "IEEE",
      "pages": "10402-10411",
      "alternate_pagination": {
        "pages": "10403-10412",
        "version": "CVF open-access copy",
        "url": "https://openaccess.thecvf.com/content_ICCV_2019/html/Liang_VrR-VG_Refocusing_Visually-Relevant_Relationships_ICCV_2019_paper.html"
      }
    },
    "identifiers": {
      "doi": "10.1109/ICCV.2019.01050",
      "arxiv": "1902.00313",
      "arxiv_primary_class": "cs.CV"
    },
    "arxiv_dates": {
      "first_posted": "2019-02-01",
      "last_revised": "2019-08-26"
    },
    "keywords": [
      "visual relationships",
      "scene graphs",
      "dataset bias",
      "Visual Genome",
      "representation learning"
    ],
    "official_abstract": "Relationships encode the interactions among individual instances and play a critical role in deep visual scene understanding. Suffering from the high predictability with non-visual information, relationship models tend to fit the statistical bias rather than \"learning\" to infer the relationships from images. To encourage further development in visual relationships, we propose a novel method to mine more valuable relationships by automatically pruning visually-irrelevant relationships. We construct a new scene graph dataset named Visually-Relevant Relationships Dataset (VrR-VG) based on Visual Genome. Compared with existing datasets, the performance gap between learnable and statistical method is more significant in VrR-VG, and frequency-based analysis does not work anymore. Moreover, we propose to learn a relationship-aware representation by jointly considering instances, attributes and relationships. By applying the representation-aware feature learned on VrR-VG, the performances of image captioning and visual question answering are systematically improved, which demonstrates the effectiveness of both our dataset and features embedding schema. Both our VrR-VG dataset and representation-aware features will be made publicly available soon.",
    "official_abstract_rights": "Excluded from the site's CC BY 4.0 license; original paper rights apply.",
    "primary_source": {
      "label": "IEEE version of record",
      "url": "https://doi.org/10.1109/ICCV.2019.01050",
      "version": "ICCV 2019 version of record; CVF open-access copy has alternate pagination 10403-10412; arXiv:1902.00313v2 cross-checked",
      "method_locator": "Abstract; visually-irrelevant relationship pruning and relationship-aware representation sections",
      "evidence_locator": "Abstract; scene-graph analysis, image captioning, and VQA experiments"
    },
    "source_checked": "2026-07-31",
    "verification_status": "author-verified",
    "author_verified_on": "2026-07-31",
    "commentary": {
      "license": "https://creativecommons.org/licenses/by/4.0/",
      "en": {
        "summary": "VrR-VG prunes relationships that can be predicted from non-visual statistics, builds a visually relevant scene-graph dataset from Visual Genome, and learns representations that jointly encode instances, attributes, and relationships.",
        "problem": "Visual-relationship models can exploit category and frequency biases to predict predicates without using image evidence, making benchmark performance a weak test of visual reasoning.",
        "contributions": [
          "Automatically identifies and removes visually irrelevant relationships from Visual Genome.",
          "Constructs the VrR-VG dataset so statistical shortcuts are less effective and visual evidence matters more.",
          "Learns a relationship-aware representation over instances, attributes, and relations for downstream tasks."
        ],
        "evidence": "The paper analyzes the gap between learnable and statistical methods on VrR-VG and reports systematic improvements in image captioning and visual question answering using the learned features. Exact margins belong to the ICCV tables.",
        "limitations": "Pruning is tied to the paper's definition and detector of visual irrelevance; some relationships can legitimately combine visual and contextual knowledge. Dataset debiasing does not eliminate every possible shortcut.",
        "positioning": "VrR-VG is both a dataset-refocusing and representation-learning contribution. It operationalizes a useful debiasing idea: first detect what can be guessed without pixels, then construct an evaluation set where image evidence is more necessary.",
        "citation_ready": "Liang et al. construct VrR-VG by pruning visually irrelevant relationships from Visual Genome and learn relationship-aware features that jointly model instances, attributes, and relations."
      },
      "zh-CN": {
        "summary": "VrR-VG 剪除仅凭非视觉统计就能预测的关系，从 Visual Genome 构建更强调视觉证据的 scene-graph 数据集，并联合编码实例、属性与关系来学习表示。",
        "problem": "视觉关系模型可能利用类别与频率偏差，在几乎不看图像证据的情况下预测 predicate，使 benchmark 分数无法真实反映视觉推理。",
        "contributions": [
          "自动识别并移除 Visual Genome 中 visually irrelevant 的关系。",
          "构建 VrR-VG，使统计捷径更难奏效、视觉证据更加必要。",
          "联合实例、属性和关系学习 relationship-aware representation，并用于下游任务。"
        ],
        "evidence": "论文分析 VrR-VG 上 learnable method 与 statistical method 的差异，并报告学习特征对 image captioning 和 visual question answering 的系统性改善；准确幅度应引用 ICCV 表格。",
        "limitations": "剪除过程依赖论文对视觉无关性的定义和检测方式，一些关系本来就会结合视觉与上下文知识；数据集去偏也不能消除所有潜在 shortcut。",
        "positioning": "VrR-VG 同时贡献数据集重构与关系表示学习。其通用去偏思路是：先检测哪些答案不用像素也能猜到，再构造更需要视觉证据的评测数据。",
        "citation_ready": "Liang 等从 Visual Genome 中剪除 visually irrelevant relationships 构建 VrR-VG，并联合建模实例、属性和关系来学习 relationship-aware feature。"
      }
    },
    "citation": {
      "id": "liang2019vrrvg",
      "type": "paper-conference",
      "title": "VrR-VG: Refocusing Visually-Relevant Relationships",
      "author": [
        {
          "given": "Yuanzhi",
          "family": "Liang"
        },
        {
          "given": "Yalong",
          "family": "Bai"
        },
        {
          "given": "Wei",
          "family": "Zhang"
        },
        {
          "given": "Xueming",
          "family": "Qian"
        },
        {
          "given": "Li",
          "family": "Zhu"
        },
        {
          "given": "Tao",
          "family": "Mei"
        }
      ],
      "container-title": "2019 IEEE/CVF International Conference on Computer Vision (ICCV)",
      "issued": {
        "date-parts": [
          [
            2019
          ]
        ]
      },
      "URL": "https://doi.org/10.1109/ICCV.2019.01050",
      "abstract": "Relationships encode the interactions among individual instances and play a critical role in deep visual scene understanding. Suffering from the high predictability with non-visual information, relationship models tend to fit the statistical bias rather than \"learning\" to infer the relationships from images. To encourage further development in visual relationships, we propose a novel method to mine more valuable relationships by automatically pruning visually-irrelevant relationships. We construct a new scene graph dataset named Visually-Relevant Relationships Dataset (VrR-VG) based on Visual Genome. Compared with existing datasets, the performance gap between learnable and statistical method is more significant in VrR-VG, and frequency-based analysis does not work anymore. Moreover, we propose to learn a relationship-aware representation by jointly considering instances, attributes and relationships. By applying the representation-aware feature learned on VrR-VG, the performances of image captioning and visual question answering are systematically improved, which demonstrates the effectiveness of both our dataset and features embedding schema. Both our VrR-VG dataset and representation-aware features will be made publicly available soon.",
      "keyword": "visual relationships, scene graphs, dataset bias, Visual Genome, representation learning",
      "publisher": "IEEE",
      "DOI": "10.1109/ICCV.2019.01050",
      "page": "10402-10411"
    },
    "resources": [
      {
        "label": "Version of record",
        "url": "https://doi.org/10.1109/ICCV.2019.01050"
      },
      {
        "label": "Open access",
        "url": "https://openaccess.thecvf.com/content_ICCV_2019/html/Liang_VrR-VG_Refocusing_Visually-Relevant_Relationships_ICCV_2019_paper.html"
      },
      {
        "label": "arXiv",
        "url": "https://arxiv.org/abs/1902.00313"
      },
      {
        "label": "PDF",
        "url": "https://openaccess.thecvf.com/content_ICCV_2019/papers/Liang_VrR-VG_Refocusing_Visually-Relevant_Relationships_ICCV_2019_paper.pdf"
      }
    ]
  }
]
