[
  {
    "id": "liang2026embodiedbrains",
    "type": "article",
    "title": "From World Action Models to Embodied Brains: A Roadmap for Open-World Physical Intelligence",
    "author": [
      {
        "given": "Yuanzhi",
        "family": "Liang"
      },
      {
        "given": "Xufeng",
        "family": "Zhan"
      },
      {
        "given": "Haibin",
        "family": "Huang"
      },
      {
        "given": "Chi",
        "family": "Zhang"
      },
      {
        "given": "Xuelong",
        "family": "Li"
      }
    ],
    "container-title": "arXiv",
    "issued": {
      "date-parts": [
        [
          2026,
          7,
          13
        ]
      ]
    },
    "URL": "https://arxiv.org/abs/2607.11689",
    "abstract": "Artificial general intelligence ultimately requires agents that can reason and act in the physical world. Action models, vision-language-action policies, and world models have advanced this goal, while World Action Models (WAMs) are particularly promising because they connect candidate interventions with predicted consequences. However, progress remains fragmented: models use incompatible action spaces and prediction targets, datasets and tasks follow different conventions, and runtime systems expose limited interfaces for reuse and evaluation. We review the evolution toward WAMs and organize these limitations into three coupled gaps: model roles and representations, objectives and standardization, and system composition. Building on this analysis, we propose a co-evolution roadmap for physical intelligence centered on the embodied brain, a long-term model target for integrating multimodal context, comparing candidate interventions, and issuing state-transition or capability requests rather than direct actuator commands. WAMs provide promising prototypes for its predictive functions, while a physical harness grounds model outputs through tools, controllers, verification, and trace logging. Shared contracts align heterogeneous models, data, tasks, and embodiments, and closed-loop post-training converts verified interaction into reusable experience. Together, these components define a modular physical-intelligence stack for adaptive and self-improving embodied agents.",
    "keyword": "world action models, embodied intelligence, physical intelligence, world models, robotics",
    "archive": "arXiv",
    "archive_location": "2607.11689",
    "genre": "Preprint"
  },
  {
    "id": "liang2026teleboost",
    "type": "article",
    "title": "TeleBoost: A Systematic Alignment Framework for High-Fidelity, Controllable, and Robust Video Generation",
    "author": [
      {
        "given": "Yuanzhi",
        "family": "Liang"
      },
      {
        "given": "Xuan'er",
        "family": "Wu"
      },
      {
        "given": "Yirui",
        "family": "Liu"
      },
      {
        "given": "Yijie",
        "family": "Fang"
      },
      {
        "given": "Yizhen",
        "family": "Fan"
      },
      {
        "given": "Ke",
        "family": "Hao"
      },
      {
        "given": "Rui",
        "family": "Li"
      },
      {
        "given": "Ruiying",
        "family": "Liu"
      },
      {
        "given": "Ziqi",
        "family": "Ni"
      },
      {
        "given": "Peng",
        "family": "Yu"
      },
      {
        "given": "Yanbo",
        "family": "Wang"
      },
      {
        "given": "Haibin",
        "family": "Huang"
      },
      {
        "given": "Qizhen",
        "family": "Weng"
      },
      {
        "given": "Chi",
        "family": "Zhang"
      },
      {
        "given": "Xuelong",
        "family": "Li"
      }
    ],
    "container-title": "arXiv",
    "issued": {
      "date-parts": [
        [
          2026,
          2,
          7
        ]
      ]
    },
    "URL": "https://arxiv.org/abs/2602.07595",
    "abstract": "Post-training is the decisive step for converting a pretrained video generator into a production-oriented model that is instruction-following, controllable, and robust over long temporal horizons. This report presents a systematical post-training framework that organizes supervised policy shaping, reward-driven reinforcement learning, and preference-based refinement into a single stability-constrained optimization stack. The framework is designed around practical video-generation constraints, including high rollout cost, temporally compounding failure modes, and feedback that is heterogeneous, uncertain, and often weakly discriminative. By treating optimization as a staged, diagnostic-driven process rather than a collection of isolated tricks, the report summarizes a cohesive recipe for improving perceptual fidelity, temporal coherence, and prompt adherence while preserving the controllability established at initialization. The resulting framework provides a clear blueprint for building scalable post-training pipelines that remain stable, extensible, and effective in real-world deployment settings.",
    "keyword": "video generation, post-training, alignment, reinforcement learning, preference optimization",
    "archive": "arXiv",
    "archive_location": "2602.07595",
    "genre": "Preprint"
  },
  {
    "id": "liang2026rlvisualgeneration",
    "type": "article-journal",
    "title": "Integrating reinforcement learning with visual generative models: foundations and advances",
    "author": [
      {
        "given": "Yuanzhi",
        "family": "Liang"
      },
      {
        "given": "Yijie",
        "family": "Fang"
      },
      {
        "given": "Rui",
        "family": "Li"
      },
      {
        "given": "Ziqi",
        "family": "Ni"
      },
      {
        "given": "Ruijie",
        "family": "Su"
      },
      {
        "given": "Chi",
        "family": "Zhang"
      }
    ],
    "container-title": "Vicinagearth",
    "issued": {
      "date-parts": [
        [
          2026,
          1,
          29
        ]
      ]
    },
    "URL": "https://doi.org/10.1007/s44336-025-00030-z",
    "abstract": "Generative models have made significant progress in synthesizing visual content, including images, videos, and 3D/4D structures. However, they are typically trained with surrogate objectives such as likelihood or reconstruction loss, which often misalign with perceptual quality, semantic accuracy, or physical realism. Reinforcement learning (RL) offers a principled framework for optimizing non-differentiable, preference-driven, and temporally structured objectives. Recent advances demonstrate its effectiveness in enhancing controllability, consistency, and human alignment across generative tasks. This survey provides a systematic overview of RL-based methods for visual content generation. We review the evolution of RL from classical control to its role as a general-purpose optimization tool, and examine its integration into image, video, and 3D/4D generation. Across these domains, RL serves not only as a fine-tuning mechanism but also as a structural component for aligning generation with complex, high-level goals. We conclude with open challenges and future research directions at the intersection of RL and generative modeling.",
    "keyword": "reinforcement learning, visual generative models, image generation, video generation, 3D and 4D generation, survey",
    "publisher": "Springer Nature",
    "DOI": "10.1007/s44336-025-00030-z",
    "volume": "3",
    "issue": "1",
    "page": "2"
  },
  {
    "id": "ni2026vipo",
    "type": "paper-conference",
    "title": "Seeing What Matters: Visual Preference Policy Optimization for Visual Generation",
    "author": [
      {
        "given": "Ziqi",
        "family": "Ni"
      },
      {
        "given": "Yuanzhi",
        "family": "Liang"
      },
      {
        "given": "Rui",
        "family": "Li"
      },
      {
        "given": "Yi",
        "family": "Zhou"
      },
      {
        "given": "Haibin",
        "family": "Huang"
      },
      {
        "given": "Chi",
        "family": "Zhang"
      },
      {
        "given": "Xuelong",
        "family": "Li"
      }
    ],
    "container-title": "Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition",
    "issued": {
      "date-parts": [
        [
          2026
        ]
      ]
    },
    "URL": "https://openaccess.thecvf.com/content/CVPR2026/html/Ni_Seeing_What_Matters_Visual_Preference_Policy_Optimization_for_Visual_Generation_CVPR_2026_paper.html",
    "abstract": "Reinforcement learning (RL) has become a powerful tool for post-training visual generative models, with Group Relative Policy Optimization (GRPO) increasingly used to align generators with human preferences. However, existing GRPO pipelines rely on a single scalar reward per sample, treating each image or video as a holistic entity and ignoring the rich spatial and temporal structure of visual content. This coarse supervision hinders the correction of localized artifacts and the modeling of fine-grained perceptual cues. We introduce Visual Preference Policy Optimization (ViPO), a GRPO variant that lifts scalar feedback into structured, pixel-level advantages. ViPO employs a Perceptual Structuring Module that uses pretrained vision backbones to construct spatially and temporally aware advantage maps, redistributing optimization pressure toward perceptually important regions while preserving the stability of standard GRPO. Across both image and video benchmarks, ViPO consistently outperforms vanilla GRPO, improving in-domain alignment with human-preference rewards and enhancing generalization on out-of-domain evaluations. The method is architecture-agnostic, lightweight, and fully compatible with existing GRPO training pipelines, providing a more expressive and informative learning signal for visual generation.",
    "keyword": "visual generation, GRPO, pixel-level advantage, preference optimization, structured feedback",
    "page": "27260-27269"
  },
  {
    "id": "li2026rats",
    "type": "paper-conference",
    "title": "Reward-Aware Trajectory Shaping for Few-step Visual Generation",
    "author": [
      {
        "given": "Rui",
        "family": "Li"
      },
      {
        "given": "Bingyu",
        "family": "Li"
      },
      {
        "given": "Yuanzhi",
        "family": "Liang"
      },
      {
        "given": "Haibin",
        "family": "Huang"
      },
      {
        "given": "Chi",
        "family": "Zhang"
      },
      {
        "given": "XueLong",
        "family": "Li"
      }
    ],
    "container-title": "34th ACM International Conference on Multimedia (ACM Multimedia 2026)",
    "issued": {
      "date-parts": [
        [
          2026
        ]
      ]
    },
    "URL": "https://arxiv.org/abs/2604.14910",
    "abstract": "Achieving high-fidelity generation in extremely few sampling steps has long been a central goal of generative modeling. Existing approaches largely rely on distillation-based frameworks to compress the original multi-step denoising process into a few-step generator. However, such methods inherently constrain the student to imitate a stronger multi-step teacher, imposing the teacher as an upper bound on student performance. We argue that introducing preference alignment awareness enables the student to optimize toward reward-preferred generation quality, potentially surpassing the teacher instead of being restricted to rigid teacher imitation. To this end, we propose Reward-Aware Trajectory Shaping (RATS), a lightweight framework for preference-aligned few-step generation. Specifically, teacher and student latent trajectories are aligned at key denoising stages through horizon matching, while a reward-aware gate is introduced to adaptively regulate teacher guidance based on their relative reward performance. Trajectory shaping is strengthened when the teacher achieves higher rewards, and relaxed when the student matches or surpasses the teacher, thereby enabling continued reward-driven improvement. By seamlessly integrating trajectory distillation, reward-aware gating, and preference alignment, RATS effectively transfers preference-relevant knowledge from high-step generators without incurring additional test-time computational overhead. Experimental results demonstrate that RATS substantially improves the efficiency--quality trade-off in few-step visual generation, significantly narrowing the gap between few-step students and stronger multi-step generators.",
    "keyword": "few-step generation, trajectory distillation, preference alignment, reward-aware gating, diffusion models",
    "publisher": "ACM",
    "archive": "arXiv",
    "archive_location": "2604.14910",
    "genre": "Forthcoming conference paper",
    "status": "forthcoming"
  },
  {
    "id": "li2026taros",
    "type": "paper-conference",
    "title": "Rethinking Reward Signals in Video GRPO: When Scores Become Targets",
    "author": [
      {
        "given": "Rui",
        "family": "Li"
      },
      {
        "given": "Yuanzhi",
        "family": "Liang"
      },
      {
        "given": "Ziqi",
        "family": "Ni"
      },
      {
        "given": "Haibin",
        "family": "Huang"
      },
      {
        "given": "Chi",
        "family": "Zhang"
      },
      {
        "given": "Xuelong",
        "family": "Li"
      }
    ],
    "container-title": "European Conference on Computer Vision (ECCV 2026)",
    "issued": {
      "date-parts": [
        [
          2026
        ]
      ]
    },
    "URL": "https://arxiv.org/abs/2511.19356",
    "abstract": "Group Relative Policy Optimization (GRPO) enables stable and preference-oriented updates via group-wise comparisons for post-training video generation. However, GRPO directly optimizes reward-induced advantages. Under sustained optimization, the reward score can lose fidelity as a proxy for true video quality, consistent with the phenomenon described by Goodhart's Law. This leads to two recurring issues: (i) shortcut-driven optimization under composite objectives and (ii) reward saturation within prompt groups. To address these issues, we introduce TaRoS, a Target-Robust Reward Signaling framework for Video generation GRPO. TaRoS leverages component level performance assessment together with intra-group sparsity to organize multi-aspect rewards towards optimization objectives. In addition, it adaptively downweights components that exhibit saturation, thereby preserving effective optimization directions and mitigating redundancy. This maintains meaningful optimization directions and preserves within-group ranking separation, thereby preventing reward hacking and leading to more reliable policy updates. Extensive experiments show consistent improvements in visual fidelity, motion coherence, and text-video alignment over strong baselines.",
    "keyword": "video generation, GRPO, reward saturation, reward hacking, Goodhart's law",
    "archive": "arXiv",
    "archive_location": "2511.19356",
    "genre": "Forthcoming conference paper",
    "status": "forthcoming"
  },
  {
    "id": "li2026otca",
    "type": "paper-conference",
    "title": "Learning to Credit the Right Steps: Objective-aware Process Optimization for Visual Generation",
    "author": [
      {
        "given": "Rui",
        "family": "Li"
      },
      {
        "given": "Ke",
        "family": "Hao"
      },
      {
        "given": "Yuanzhi",
        "family": "Liang"
      },
      {
        "given": "Haibin",
        "family": "Huang"
      },
      {
        "given": "Chi",
        "family": "Zhang"
      },
      {
        "given": "Yun",
        "family": "Gu"
      },
      {
        "given": "XueLong",
        "family": "Li"
      }
    ],
    "container-title": "34th ACM International Conference on Multimedia (ACM Multimedia 2026)",
    "issued": {
      "date-parts": [
        [
          2026
        ]
      ]
    },
    "URL": "https://arxiv.org/abs/2604.19234",
    "abstract": "Reinforcement learning, particularly Group Relative Policy Optimization (GRPO), has emerged as an effective framework for post-training visual generative models with human preference signals. However, its effectiveness is fundamentally limited by coarse reward credit assignment. In modern visual generation, multiple reward models are often used to capture heterogeneous objectives, such as visual quality, motion consistency, and text alignment. Existing GRPO pipelines typically collapse these rewards into a single static scalar and propagate it uniformly across the entire diffusion trajectory. This design ignores the stage-specific roles of different denoising steps and produces mistimed or incompatible optimization signals. To address this issue, we propose Objective-aware Trajectory Credit Assignment (OTCA), a structured framework for fine-grained GRPO training. OTCA consists of two key components. Trajectory-Level Credit Decomposition estimates the relative importance of different denoising steps. Multi-Objective Credit Allocation adaptively weights and combines multiple reward signals throughout the denoising process. By jointly modeling temporal credit and objective-level credit, OTCA converts coarse reward supervision into a structured, timestep-aware training signal that better matches the iterative nature of diffusion-based generation. Extensive experiments show that OTCA consistently improves both image and video generation quality across evaluation metrics.",
    "keyword": "visual generation, GRPO, credit assignment, multi-objective optimization, diffusion models",
    "publisher": "ACM",
    "archive": "arXiv",
    "archive_location": "2604.19234",
    "genre": "Forthcoming conference paper",
    "status": "forthcoming"
  },
  {
    "id": "liu2026bpgo",
    "type": "paper-conference",
    "title": "Learning What to Trust: Bayesian Prior-Guided Optimization for Visual Generation",
    "author": [
      {
        "given": "Ruiying",
        "family": "Liu"
      },
      {
        "given": "Yuanzhi",
        "family": "Liang"
      },
      {
        "given": "Haibin",
        "family": "Huang"
      },
      {
        "given": "Tianshu",
        "family": "Yu"
      },
      {
        "given": "Chi",
        "family": "Zhang"
      }
    ],
    "container-title": "Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition",
    "issued": {
      "date-parts": [
        [
          2026
        ]
      ]
    },
    "URL": "https://openaccess.thecvf.com/content/CVPR2026/html/Liu_Learning_What_to_Trust_Bayesian_Prior-Guided_Optimization_for_Visual_Generation_CVPR_2026_paper.html",
    "abstract": "Group Relative Policy Optimization (GRPO) has emerged as an effective and lightweight framework for post-training visual generative models. However, its performance is fundamentally limited by the ambiguity of textual visual correspondence: a single prompt may validly describe diverse visual outputs, and a single image or video may support multiple equally correct interpretations. This many to many relationship leads reward models to generate uncertain and weakly discriminative signals, causing GRPO to underutilize reliable feedback and overfit noisy ones. We introduce Bayesian Prior-Guided Optimization (BPGO), a novel extension of GRPO that explicitly models reward uncertainty through a semantic prior anchor. BPGO adaptively modulates optimization trust at two levels: inter-group Bayesian trust allocation emphasizes updates from groups consistent with the prior while down-weighting ambiguous ones, and intra-group prior-anchored renormalization sharpens sample distinctions by expanding confident deviations and compressing uncertain scores. Across both image and video generation tasks, BPGO delivers consistently stronger semantic alignment, enhanced perceptual fidelity, and faster convergence than standard GRPO and recent variants.",
    "keyword": "visual generation, GRPO, reward uncertainty, Bayesian prior, semantic alignment",
    "page": "34408-34417"
  },
  {
    "id": "liu2026laxmotion",
    "type": "paper-conference",
    "title": "LaxMotion: Rethinking Supervision Granularity for 3D Human Motion Generation",
    "author": [
      {
        "given": "Sheng",
        "family": "Liu"
      },
      {
        "given": "Yuanzhi",
        "family": "Liang"
      },
      {
        "given": "Sidan",
        "family": "Du"
      }
    ],
    "container-title": "European Conference on Computer Vision (ECCV 2026)",
    "issued": {
      "date-parts": [
        [
          2026
        ]
      ]
    },
    "URL": "https://arxiv.org/abs/2511.11368",
    "abstract": "Recent 3D human motion generation models demonstrate remarkable reconstruction accuracy yet struggle to generalize beyond training distributions. This limitation arises partly from the use of precise 3D supervision, which encourages models to fit fixed coordinate patterns instead of learning the essential 3D structure and motion semantic cues required for robust generalization. To overcome this limitation, we propose LaxMotion, a framework that synthesizes realistic 3D motions without direct 3D pose supervision. Instead of regressing toward exact coordinates, LaxMotion learns 3D motion as a consistent explanation of global trajectories and monocular 2D kinematic cues. We introduce a structured motion factorization together with a reformulated training paradigm under relaxed observability. This design is further supported by relaxed regularization objectives that enforce view consistent alignment, orientation coherence, and structural stability. Under this relaxed supervision paradigm, LaxMotion generates diverse, temporally coherent, and semantically aligned 3D motions, achieving performance comparable to or surpassing fully 3D supervised methods. These results indicate that shifting supervision from exact coordinate matching to structural consistency promotes stronger reasoning and improved generalization, offering a scalable and data efficient paradigm for 3D motion generation.",
    "keyword": "3D human motion, relaxed supervision, motion generation, structural consistency, generalization",
    "archive": "arXiv",
    "archive_location": "2511.11368",
    "genre": "Forthcoming conference paper",
    "status": "forthcoming"
  },
  {
    "id": "chen2025teleworld",
    "type": "article",
    "title": "TeleWorld: Towards Dynamic Multimodal Synthesis with a 4D World Model",
    "author": [
      {
        "given": "Yabo",
        "family": "Chen"
      },
      {
        "given": "Yuanzhi",
        "family": "Liang"
      },
      {
        "given": "Jiepeng",
        "family": "Wang"
      },
      {
        "given": "Tingxi",
        "family": "Chen"
      },
      {
        "given": "Junfei",
        "family": "Cheng"
      },
      {
        "given": "Zixiao",
        "family": "Gu"
      },
      {
        "given": "Yuyang",
        "family": "Huang"
      },
      {
        "given": "Zicheng",
        "family": "Jiang"
      },
      {
        "given": "Wei",
        "family": "Li"
      },
      {
        "given": "Tian",
        "family": "Li"
      },
      {
        "given": "Weichen",
        "family": "Li"
      },
      {
        "given": "Zuoxin",
        "family": "Li"
      },
      {
        "given": "Guangce",
        "family": "Liu"
      },
      {
        "given": "Jialun",
        "family": "Liu"
      },
      {
        "given": "Junqi",
        "family": "Liu"
      },
      {
        "given": "Haoyuan",
        "family": "Wang"
      },
      {
        "given": "Qizhen",
        "family": "Weng"
      },
      {
        "given": "Xuan'er",
        "family": "Wu"
      },
      {
        "given": "Xunzhi",
        "family": "Xiang"
      },
      {
        "given": "Xiaoyan",
        "family": "Yang"
      },
      {
        "given": "Xin",
        "family": "Zhang"
      },
      {
        "given": "Shiwen",
        "family": "Zhang"
      },
      {
        "given": "Junyu",
        "family": "Zhou"
      },
      {
        "given": "Chengcheng",
        "family": "Zhou"
      },
      {
        "given": "Haibin",
        "family": "Huang"
      },
      {
        "given": "Chi",
        "family": "Zhang"
      },
      {
        "given": "Xuelong",
        "family": "Li"
      }
    ],
    "container-title": "arXiv",
    "issued": {
      "date-parts": [
        [
          2025,
          12,
          31
        ]
      ]
    },
    "URL": "https://arxiv.org/abs/2601.00051",
    "abstract": "World models aim to endow AI systems with the ability to represent, generate, and interact with dynamic environments in a coherent and temporally consistent manner. While recent video generation models have demonstrated impressive visual quality, they remain limited in real-time interaction, long-horizon consistency, and persistent memory of dynamic scenes, hindering their evolution into practical world models. In this report, we present TeleWorld, a real-time multimodal 4D world modeling framework that unifies video generation, dynamic scene reconstruction, and long-term world memory within a closed-loop system. TeleWorld introduces a novel generation-reconstruction-guidance paradigm, where generated video streams are continuously reconstructed into a dynamic 4D spatio-temporal representation, which in turn guides subsequent generation to maintain spatial, temporal, and physical consistency. To support long-horizon generation with low latency, we employ an autoregressive diffusion-based video model enhanced with Macro-from-Micro Planning (MMPL)--a hierarchical planning method that reduces error accumulation from frame-level to segment-level-alongside efficient Distribution Matching Distillation (DMD), enabling real-time synthesis under practical computational budgets. Our approach achieves seamless integration of dynamic object modeling and static scene representation within a unified 4D framework, advancing world models toward practical, interactive, and computationally accessible systems. Extensive experiments demonstrate that TeleWorld achieves strong performance in both static and dynamic world understanding, long-term consistency, and real-time generation efficiency, positioning it as a practical step toward interactive, memory-enabled world models for multimodal generation and embodied intelligence.",
    "keyword": "4D world model, video generation, dynamic reconstruction, long-term memory, real-time synthesis",
    "archive": "arXiv",
    "archive_location": "2601.00051",
    "genre": "Preprint"
  },
  {
    "id": "liu2025uniinter",
    "type": "paper-conference",
    "title": "Uni-Inter: Unifying 3D Human Motion Synthesis Across Diverse Interaction Contexts",
    "author": [
      {
        "given": "Sheng",
        "family": "Liu"
      },
      {
        "given": "Yuanzhi",
        "family": "Liang"
      },
      {
        "given": "Jiepeng",
        "family": "Wang"
      },
      {
        "given": "Sidan",
        "family": "Du"
      },
      {
        "given": "Chi",
        "family": "Zhang"
      },
      {
        "given": "Xuelong",
        "family": "Li"
      }
    ],
    "container-title": "Proceedings of the SIGGRAPH Asia 2025 Conference Papers",
    "issued": {
      "date-parts": [
        [
          2025,
          12,
          14
        ]
      ]
    },
    "URL": "https://doi.org/10.1145/3757377.3763954",
    "abstract": "We present Uni-Inter, a unified framework for human motion generation that supports a wide range of interaction scenarios: including human-human, human-object, and human-scene-within a single, task-agnostic architecture. In contrast to existing methods that rely on task-specific designs and exhibit limited generalization, Uni-Inter introduces the Unified Interactive Volume (UIV), a volumetric representation that encodes heterogeneous interactive entities into a shared spatial field. This enables consistent relational reasoning and compound interaction modeling. Motion generation is formulated as joint-wise probabilistic prediction over the UIV, allowing the model to capture fine-grained spatial dependencies and produce coherent, context-aware behaviors. Experiments across three representative interaction tasks demonstrate that Uni-Inter achieves competitive performance and generalizes well to novel combinations of entities. These results suggest that unified modeling of compound interactions offers a promising direction for scalable motion synthesis in complex environments.",
    "keyword": "3D human motion, human-object interaction, human-human interaction, human-scene interaction, unified representation",
    "publisher": "ACM",
    "DOI": "10.1145/3757377.3763954",
    "page": "1-11"
  },
  {
    "id": "ma2025intersyn",
    "type": "paper-conference",
    "title": "InterSyn: Interleaved Learning for Dynamic Motion Synthesis in the Wild",
    "author": [
      {
        "given": "Yiyi",
        "family": "Ma"
      },
      {
        "given": "Yuanzhi",
        "family": "Liang"
      },
      {
        "given": "Xiu",
        "family": "Li"
      },
      {
        "given": "Chi",
        "family": "Zhang"
      },
      {
        "given": "Xuelong",
        "family": "Li"
      }
    ],
    "container-title": "2025 IEEE/CVF International Conference on Computer Vision (ICCV)",
    "issued": {
      "date-parts": [
        [
          2025
        ]
      ]
    },
    "URL": "https://doi.org/10.1109/ICCV51701.2025.01192",
    "abstract": "We present Interleaved Learning for Motion Synthesis (InterSyn), a novel framework that targets the generation of realistic interaction motions by learning from integrated motions that consider both solo and multi-person dynamics. Unlike previous methods that treat these components separately, InterSyn employs an interleaved learning strategy to capture the natural, dynamic interactions and nuanced coordination inherent in real-world scenarios. Our framework comprises two key modules: the Interleaved Interaction Synthesis (INS) module, which jointly models solo and interactive behaviors in a unified paradigm from a first-person perspective to support multiple character interactions, and the Relative Coordination Refinement (REC) module, which refines mutual dynamics and ensures synchronized motions among characters. Experimental results show that the motion sequences generated by InterSyn exhibit higher text-to-motion alignment and improved diversity compared with recent methods, setting a new benchmark for robust and natural motion synthesis. Additionally, our code will be open-sourced in the future to promote further research and development in this area.",
    "keyword": "3D human motion, multi-person interaction, interleaved learning, motion synthesis, coordination refinement",
    "publisher": "IEEE",
    "DOI": "10.1109/ICCV51701.2025.01192",
    "page": "12832-12841"
  },
  {
    "id": "zhang2024vast",
    "type": "article",
    "title": "VAST 1.0: A Unified Framework for Controllable and Consistent Video Generation",
    "author": [
      {
        "given": "Chi",
        "family": "Zhang"
      },
      {
        "given": "Yuanzhi",
        "family": "Liang"
      },
      {
        "given": "Xi",
        "family": "Qiu"
      },
      {
        "given": "Fangqiu",
        "family": "Yi"
      },
      {
        "given": "Xuelong",
        "family": "Li"
      }
    ],
    "container-title": "arXiv",
    "issued": {
      "date-parts": [
        [
          2024,
          12,
          21
        ]
      ]
    },
    "URL": "https://arxiv.org/abs/2412.16677",
    "abstract": "Generating high-quality videos from textual descriptions poses challenges in maintaining temporal coherence and control over subject motion. We propose VAST (Video As Storyboard from Text), a two-stage framework to address these challenges and enable high-quality video generation. In the first stage, StoryForge transforms textual descriptions into detailed storyboards, capturing human poses and object layouts to represent the structural essence of the scene. In the second stage, VisionForge generates videos from these storyboards, producing high-quality videos with smooth motion, temporal consistency, and spatial coherence. By decoupling text understanding from video generation, VAST enables precise control over subject dynamics and scene composition. Experiments on the VBench benchmark demonstrate that VAST outperforms existing methods in both visual quality and semantic expression, setting a new standard for dynamic and coherent video generation.",
    "keyword": "video generation, storyboard, controllable generation, temporal consistency, scene composition",
    "archive": "arXiv",
    "archive_location": "2412.16677",
    "genre": "Preprint"
  },
  {
    "id": "liang2024mhem",
    "type": "article-journal",
    "title": "Penalizing the Hard Example But Not Too Much: A Strong Baseline for Fine-Grained Visual Classification",
    "author": [
      {
        "given": "Yuanzhi",
        "family": "Liang"
      },
      {
        "given": "Linchao",
        "family": "Zhu"
      },
      {
        "given": "Xiaohan",
        "family": "Wang"
      },
      {
        "given": "Yi",
        "family": "Yang"
      }
    ],
    "container-title": "IEEE Transactions on Neural Networks and Learning Systems",
    "issued": {
      "date-parts": [
        [
          2024,
          5
        ]
      ]
    },
    "URL": "https://doi.org/10.1109/TNNLS.2022.3213563",
    "abstract": "Though significant progress has been achieved on fine-grained visual classification (FGVC), severe overfitting still hinders model generalization. A recent study shows that hard samples in the training set can be easily fitted, but most existing FGVC methods fail to classify some hard examples in the test set. The reason is that the model overfits those hard examples in the training set, but does not learn to generalize to unseen examples in the test set. In this paper, we propose a Moderate Hard Example Modulation (MHEM) strategy to properly modulate the hard examples. MHEM encourages the model to not overfit hard examples and offers better generalization and discrimination. First, we introduce three conditions and formulate a general form of a modulated loss function. Second, we instantiate the loss function and provide a strong baseline for FGVC, where the performance of a naive backbone can be boosted and be comparable with recent methods. Moreover, we demonstrate that our baseline can be readily incorporated into the existing methods and empower these methods to be more discriminative. Equipped with our strong baseline, we achieve consistent improvements on three typical fine-grained visual classification datasets, i.e., CUB-200-2011, Stanford Cars, and FGVC-Aircraft. We hope the idea of Moderate Hard Example Modulation will inspire future research work toward more effective fine-grained visual recognition.",
    "keyword": "fine-grained visual classification, hard examples, loss modulation, generalization, overfitting",
    "publisher": "IEEE",
    "DOI": "10.1109/TNNLS.2022.3213563",
    "volume": "35",
    "issue": "5",
    "page": "7048-7059"
  },
  {
    "id": "liang2024anteval",
    "type": "article",
    "title": "AntEval: Evaluation of Social Interaction Competencies in LLM-Driven Agents",
    "author": [
      {
        "given": "Yuanzhi",
        "family": "Liang"
      },
      {
        "given": "Linchao",
        "family": "Zhu"
      },
      {
        "given": "Yi",
        "family": "Yang"
      }
    ],
    "container-title": "arXiv",
    "issued": {
      "date-parts": [
        [
          2024,
          1,
          12
        ]
      ]
    },
    "URL": "https://arxiv.org/abs/2401.06509",
    "abstract": "Large Language Models (LLMs) have demonstrated their ability to replicate human behaviors across a wide range of scenarios. However, their capability in handling complex, multi-character social interactions has yet to be fully explored, primarily due to the absence of robust, quantitative evaluation methods. This gap has slowed the development of agents proficient in more nuanced interactions beyond simple exchanges, for example, small talk. To address this challenge, we introduce the Multi-Agent Interaction Evaluation Framework (AntEval), encompassing a novel interaction framework and evaluation methods. The interaction framework aims to foster an complex interaction environment that bolsters information exchange and intention expression within social interactions. Furthermore, we introduce evaluation methods, including two metrics: Information Exchanging Precision (IEP) and Interaction Expressiveness Gap (IEG), designed for the quantitative and objective assessment of agents' interaction competencies. Our findings highlight the utility of these evaluative methods and show significant potential for improving LLMs' ability to construct agents that interact in a more natural manner with human-like intricacy.",
    "keyword": "LLM agents, multi-agent interaction, social interaction, evaluation, information exchange",
    "archive": "arXiv",
    "archive_location": "2401.06509",
    "genre": "Preprint"
  },
  {
    "id": "liang2024icocap",
    "type": "article-journal",
    "title": "IcoCap: Improving Video Captioning by Compounding Images",
    "author": [
      {
        "given": "Yuanzhi",
        "family": "Liang"
      },
      {
        "given": "Linchao",
        "family": "Zhu"
      },
      {
        "given": "Xiaohan",
        "family": "Wang"
      },
      {
        "given": "Yi",
        "family": "Yang"
      }
    ],
    "container-title": "IEEE Transactions on Multimedia",
    "issued": {
      "date-parts": [
        [
          2024
        ]
      ]
    },
    "URL": "https://doi.org/10.1109/TMM.2023.3322329",
    "abstract": "Video captioning is a more challenging task compared to image captioning, primarily due to differences in content density. Video data contains redundant visual content, making it difficult for captioners to generalize diverse content and avoid being misled by irrelevant elements. Moreover, redundant content is not well-trimmed to match the corresponding visual semantics in the ground truth, further increasing the difficulty of video captioning. Current research in video captioning predominantly focuses on captioner design, neglecting the impact of content density on captioner performance. Considering the differences between videos and images, there exists an another line to improve video captioning by leveraging concise and easily-learned image samples to further diversify video samples. This modification to content density compels the captioner to learn more effectively against redundancy and ambiguity. In this paper, we propose a novel approach called Image-Compounded learning for video Captioners (IcoCap) to facilitate better learning of complex video semantics. IcoCap comprises two components: the Image-Video Compounding Strategy (ICS) and Visual-Semantic Guided Captioning (VGC). ICS compounds easily-learned image semantics into video semantics, further diversifying video content and prompting the network to generalize contents in a more diverse sample. Besides, learning with the sample compounded with image contents, the captioner is compelled to better extract valuable video cues in the presence of straightforward image semantics. This helps the captioner further focus on relevant information while filtering out extraneous content. Then, VGC guides the network in flexibly learning ground truth captions based on the compounded samples, helping to mitigate the mismatch between the ground truth and ambiguous semantics in video samples. Our experimental results demonstrate the effectiveness of IcoCap in improving the learning of video captioners. Applied to the widely-used MSVD, MSR-VTT, and VATEX datasets, our approach achieves competitive or superior results compared to state-of-the-art methods, illustrating its capacity to handle redundant and ambiguous video data.",
    "keyword": "video captioning, image-video compounding, content density, visual-semantic guidance, multimodal learning",
    "publisher": "IEEE",
    "DOI": "10.1109/TMM.2023.3322329",
    "volume": "26",
    "page": "4389-4400"
  },
  {
    "id": "lu2024freelong",
    "type": "paper-conference",
    "title": "FreeLong: Training-Free Long Video Generation with SpectralBlend Temporal Attention",
    "author": [
      {
        "given": "Yu",
        "family": "Lu"
      },
      {
        "given": "Yuanzhi",
        "family": "Liang"
      },
      {
        "given": "Linchao",
        "family": "Zhu"
      },
      {
        "given": "Yi",
        "family": "Yang"
      }
    ],
    "container-title": "Advances in Neural Information Processing Systems",
    "issued": {
      "date-parts": [
        [
          2024
        ]
      ]
    },
    "URL": "https://doi.org/10.52202/079017-4177",
    "abstract": "Video diffusion models have made substantial progress in various video generation applications. However, training models for long video generation tasks require significant computational and data resources, posing a challenge to developing long video diffusion models. This paper investigates a straightforward and training-free approach to extend an existing short video diffusion model (e.g. pre-trained on 16-frame videos) for consistent long video generation (e.g. 128 frames). Our preliminary observation has found that directly applying the short video diffusion model to generate long videos can lead to severe video quality degradation. Further investigation reveals that this degradation is primarily due to the distortion of high-frequency components in long videos, characterized by a decrease in spatial high-frequency components and an increase in temporal high-frequency components. Motivated by this, we propose a novel solution named FreeLong to balance the frequency distribution of long video features during the denoising process. FreeLong blends the low-frequency components of global video features, which encapsulate the entire video sequence, with the high-frequency components of local video features that focus on shorter subsequences of frames. This approach maintains global consistency while incorporating diverse and high-quality spatiotemporal details from local videos, enhancing both the consistency and fidelity of long video generation. We evaluated FreeLong on multiple base video diffusion models and observed significant improvements. Additionally, our method supports coherent multi-prompt generation, ensuring both visual coherence and seamless transitions between scenes.",
    "keyword": "long video generation, training-free, video diffusion, frequency decomposition, temporal attention",
    "editor": [
      {
        "given": "A.",
        "family": "Globerson"
      },
      {
        "given": "L.",
        "family": "Mackey"
      },
      {
        "given": "D.",
        "family": "Belgrave"
      },
      {
        "given": "A.",
        "family": "Fan"
      },
      {
        "given": "U.",
        "family": "Paquet"
      },
      {
        "given": "J.",
        "family": "Tomczak"
      },
      {
        "given": "C.",
        "family": "Zhang"
      }
    ],
    "publisher": "Curran Associates, Inc.",
    "DOI": "10.52202/079017-4177",
    "volume": "37",
    "page": "131434-131455"
  },
  {
    "id": "liang2023maal",
    "type": "paper-conference",
    "title": "MAAL: Multimodality-Aware Autoencoder-based Affordance Learning for 3D Articulated Objects",
    "author": [
      {
        "given": "Yuanzhi",
        "family": "Liang"
      },
      {
        "given": "Xiaohan",
        "family": "Wang"
      },
      {
        "given": "Linchao",
        "family": "Zhu"
      },
      {
        "given": "Yi",
        "family": "Yang"
      }
    ],
    "container-title": "2023 IEEE/CVF International Conference on Computer Vision (ICCV)",
    "issued": {
      "date-parts": [
        [
          2023
        ]
      ]
    },
    "URL": "https://doi.org/10.1109/ICCV51070.2023.00027",
    "abstract": "Inferring affordance for 3D articulated objects is a challenging and practical problem. It is a primary problem for applying robots to real-world scenarios. The exploration can be summarized as figuring out where to act and how to act. Correspondingly, the task mainly requires producing actionability scores, action proposals, and success likelihood scores according to the given 3D object information and robotic information. Current works usually directly process multi-modal inputs with early fusion and apply critic networks to produce scores, which leads to insufficient multi-modal learning ability and inefficiently iterative training in multiple stages. This paper proposes a novel Multimodality-Aware Autoencoder-based affordance Learning (MAAL) for the 3D object affordance problem. It is an efficient pipeline, trained in one go, and only requires a few positive samples in training data. More importantly, MAAL contains a MultiModal Energized Encoder (MME) for better multi-modal learning. It comprehensively models all multi-modal inputs from 3D objects and robotic actions. Jointly considering information from multiple modalities, the encoder further learns interactions between robots and objects. MME empowers the better multi-modal learning ability for understanding object affordance. Experimental results and visualizations, based on a large-scale dataset PartNet-Mobility, show the effectiveness of MAAL in learning multi-modal data and solving the 3D articulated object affordance problem.",
    "keyword": "3D affordance learning, articulated objects, multimodal learning, autoencoder, robotic interaction",
    "publisher": "IEEE",
    "DOI": "10.1109/ICCV51070.2023.00027",
    "page": "217-227"
  },
  {
    "id": "liang2022seeg",
    "type": "paper-conference",
    "title": "SEEG: Semantic Energized Co-speech Gesture Generation",
    "author": [
      {
        "given": "Yuanzhi",
        "family": "Liang"
      },
      {
        "given": "Qianyu",
        "family": "Feng"
      },
      {
        "given": "Linchao",
        "family": "Zhu"
      },
      {
        "given": "Li",
        "family": "Hu"
      },
      {
        "given": "Pan",
        "family": "Pan"
      },
      {
        "given": "Yi",
        "family": "Yang"
      }
    ],
    "container-title": "2022 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)",
    "issued": {
      "date-parts": [
        [
          2022
        ]
      ]
    },
    "URL": "https://doi.org/10.1109/CVPR52688.2022.01022",
    "abstract": "Talking gesture generation is a practical yet challenging task which aims to synthesize gestures in line with speech. Gestures with meaningful signs can better convey useful information and arouse sympathy in the audience. Current works focus on aligning gestures with the speech rhythms, which are hard to mine the semantics and model semantic gestures explicitly. In this paper, we propose a novel method SEmantic Energized Generation (SEEG), for semantic-aware gesture generation. Our method contains two parts: DEcoupled Mining module (DEM) and Semantic Energizing Module (SEM). DEM decouples the semantic-irrelevant information from inputs and separately mines information for the beat and semantic gestures. SEM conducts semantic learning and produces semantic gestures. Apart from representational similarity, SEM requires the predictions to express the same semantics as the ground truth. Besides, a semantic prompter is designed in SEM to leverage the semantic-aware supervision to predictions. This promotes the networks to learn and generate semantic gestures. Experimental results reported in three metrics on different benchmarks prove that SEEG efficiently mines semantic cues and generates semantic gestures. In comparison, SEEG outperforms other methods in all semantic-aware evaluations on different datasets. Qualitative evaluations also indicate the superiority of SEEG in semantic expressiveness.",
    "keyword": "co-speech gesture, gesture generation, semantic gestures, speech rhythm, disentangled learning",
    "publisher": "IEEE",
    "DOI": "10.1109/CVPR52688.2022.01022",
    "page": "10463-10472"
  },
  {
    "id": "liang2022elp",
    "type": "paper-conference",
    "title": "A Simple Episodic Linear Probe Improves Visual Recognition in the Wild",
    "author": [
      {
        "given": "Yuanzhi",
        "family": "Liang"
      },
      {
        "given": "Linchao",
        "family": "Zhu"
      },
      {
        "given": "Xiaohan",
        "family": "Wang"
      },
      {
        "given": "Yi",
        "family": "Yang"
      }
    ],
    "container-title": "2022 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)",
    "issued": {
      "date-parts": [
        [
          2022
        ]
      ]
    },
    "URL": "https://doi.org/10.1109/CVPR52688.2022.00934",
    "abstract": "Understanding network generalization and feature discrimination is an open research problem in visual recognition. Many studies have been conducted to assess the quality of feature representations. One of the simple strategies is to utilize a linear probing classifier to quantitatively evaluate the class accuracy under the obtained features. The typical linear probe is only applied as a proxy at the inference time, but its efficacy in measuring features' suitability for linear classification is largely neglected in training. In this paper, we propose an episodic linear probing (ELP) classifier to reflect the generalization of visual representations in an online manner. ELP is trained with detached features from the network and re-initialized episodically. It demonstrates the discriminability of the visual representations in training. Then, an ELP-suitable Regularization term (ELP-SR) is introduced to reflect the distances of probability distributions between ELP classifier and the main classifier. ELP-SR leverages a re-scaling factor to regularize each sample in training, which modulates the loss function adaptively and encourages the features to be discriminative and generalized. We observe significant improvements in three real-world visual recognition tasks, including fine-grained visual classification, long-tailed visual recognition, and generic object recognition. The performance gains show the effectiveness of our method in improving network generalization and feature discrimination.",
    "keyword": "visual recognition, generalization, linear probing, representation learning, adaptive regularization",
    "publisher": "IEEE",
    "DOI": "10.1109/CVPR52688.2022.00934",
    "page": "9549-9559"
  },
  {
    "id": "liu2021foodingredient",
    "type": "article-journal",
    "title": "Food and Ingredient Joint Learning for Fine-Grained Recognition",
    "author": [
      {
        "given": "Chengxu",
        "family": "Liu"
      },
      {
        "given": "Yuanzhi",
        "family": "Liang"
      },
      {
        "given": "Yao",
        "family": "Xue"
      },
      {
        "given": "Xueming",
        "family": "Qian"
      },
      {
        "given": "Jianlong",
        "family": "Fu"
      }
    ],
    "container-title": "IEEE Transactions on Circuits and Systems for Video Technology",
    "issued": {
      "date-parts": [
        [
          2021,
          6
        ]
      ]
    },
    "URL": "https://doi.org/10.1109/TCSVT.2020.3020079",
    "abstract": "Fine-grained food recognition is the detailed classification that provides more specialized and professional attribute information of food. It is the basic work to realize healthy diet recommendations and cooking instructions, nutrition intake management, and cafeteria self-checkout system. Chinese food lacks structured information, and ingredients composition is an important consideration. The current approaches mostly focus on global dish appearance without any analysis of ingredient composition and fully considering the attention of regional features. In this paper, we propose an Attention Fusion Network (AFN) and Food-Ingredient Joint Learning module for fine-grained food and ingredients recognition. The AFN first focuses on the food discrimination region against unstructured defeat and generates the feature embeddings jointly aware of the ingredients and food. The Food-Ingredient Joint Learning module aims at alleviating the issue of ingredients imbalance. Therefore, we propose a balance focal loss to optimize the feature expression ability of the network for ingredients. In experiments, the results of ingredients recognition show the state-of-the-art performances on fine-grained Chinese food dataset VIREO Food-172.",
    "keyword": "fine-grained food recognition, ingredient recognition, attention fusion, joint learning, class imbalance",
    "publisher": "IEEE",
    "DOI": "10.1109/TCSVT.2020.3020079",
    "volume": "31",
    "issue": "6",
    "page": "2480-2493"
  },
  {
    "id": "quan2021rain",
    "type": "paper-conference",
    "title": "Removing Raindrops and Rain Streaks in One Go",
    "author": [
      {
        "given": "Ruijie",
        "family": "Quan"
      },
      {
        "given": "Xin",
        "family": "Yu"
      },
      {
        "given": "Yuanzhi",
        "family": "Liang"
      },
      {
        "given": "Yi",
        "family": "Yang"
      }
    ],
    "container-title": "2021 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)",
    "issued": {
      "date-parts": [
        [
          2021
        ]
      ]
    },
    "URL": "https://doi.org/10.1109/CVPR46437.2021.00903",
    "abstract": "Existing rain-removal algorithms often tackle either rain streak removal or raindrop removal, and thus may fail to handle real-world rainy scenes. Besides, the lack of real-world deraining datasets comprising different types of rain and their corresponding rain-free ground-truth also impedes deraining algorithm development. In this paper, we aim to address real-world deraining problems from two aspects. First, we propose a complementary cascaded network architecture, namely CCN, to remove rain streaks and raindrops in a unified framework. Specifically, our CCN removes raindrops and rain streaks in a complementary fashion, i.e., raindrop removal followed by rain streak removal and vice versa, and then fuses the results via an attention based fusion module. Considering significant shape and structure differences between rain streaks and raindrops, it is difficult to manually design a sophisticated network to remove them effectively. Thus, we employ neural architecture search to adaptively find optimal architectures within our specified deraining search space. Second, we present a new real-world rain dataset, namely RainDS, to prosper the development of deraining algorithms in practical scenarios. RainDS consists of rain images in different types and their corresponding rain-free ground-truth, including rain streak only, raindrop only, and both of them. Extensive experimental results on both existing benchmarks and RainDS demonstrate that our method outperforms the state-of-the-art.",
    "keyword": "image deraining, raindrop removal, rain streak removal, neural architecture search, RainDS",
    "publisher": "IEEE",
    "DOI": "10.1109/CVPR46437.2021.00903",
    "page": "9143-9152"
  },
  {
    "id": "liang2019vrrvg",
    "type": "paper-conference",
    "title": "VrR-VG: Refocusing Visually-Relevant Relationships",
    "author": [
      {
        "given": "Yuanzhi",
        "family": "Liang"
      },
      {
        "given": "Yalong",
        "family": "Bai"
      },
      {
        "given": "Wei",
        "family": "Zhang"
      },
      {
        "given": "Xueming",
        "family": "Qian"
      },
      {
        "given": "Li",
        "family": "Zhu"
      },
      {
        "given": "Tao",
        "family": "Mei"
      }
    ],
    "container-title": "2019 IEEE/CVF International Conference on Computer Vision (ICCV)",
    "issued": {
      "date-parts": [
        [
          2019
        ]
      ]
    },
    "URL": "https://doi.org/10.1109/ICCV.2019.01050",
    "abstract": "Relationships encode the interactions among individual instances and play a critical role in deep visual scene understanding. Suffering from the high predictability with non-visual information, relationship models tend to fit the statistical bias rather than \"learning\" to infer the relationships from images. To encourage further development in visual relationships, we propose a novel method to mine more valuable relationships by automatically pruning visually-irrelevant relationships. We construct a new scene graph dataset named Visually-Relevant Relationships Dataset (VrR-VG) based on Visual Genome. Compared with existing datasets, the performance gap between learnable and statistical method is more significant in VrR-VG, and frequency-based analysis does not work anymore. Moreover, we propose to learn a relationship-aware representation by jointly considering instances, attributes and relationships. By applying the representation-aware feature learned on VrR-VG, the performances of image captioning and visual question answering are systematically improved, which demonstrates the effectiveness of both our dataset and features embedding schema. Both our VrR-VG dataset and representation-aware features will be made publicly available soon.",
    "keyword": "visual relationships, scene graphs, dataset bias, Visual Genome, representation learning",
    "publisher": "IEEE",
    "DOI": "10.1109/ICCV.2019.01050",
    "page": "10402-10411"
  }
]
