[
  {
    "id": "2609.10540",
    "title": "Programmable World Model",
    "authors": "Zheng-Hui Huang; Guixu Lin; Jiacheng Lin; Yi-Chuan Huang; Ruihan Yu; Muyao Niu; Siqi Yang; Yu-Lun Liu; Yung-Yu Chuang; Kaipeng Zhang; Zhixiang Wang",
    "affiliations": "Alaya Lab",
    "contribution": "Programmable World Model makes a rule-executing engine authoritative for world facts and uses a conditioned video model to render them. State-augmented 3D oriented bounding boxes connect these components without requiring detailed animated assets. On CombatStateBench, the system reports 94% visible-alive-count accuracy and 98% death-state accuracy, but these permissive global metrics do not establish correct entity-specific interactions or autonomous action selection (e-world, e-controls, e-table, e-metrics).",
    "abstract": "",
    "submittedDate": "2026-09-09",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv260910540",
    "arxivUrl": "https://arxiv.org/abs/2609.10540",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2609.10540",
    "pdfUrl": "https://arxiv.org/pdf/2609.10540",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "神经世界模拟器"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": null
  },
  {
    "id": "2609.10506",
    "title": "DUET-DINO: Simultaneous Cross-View World Modeling for Latent Planning in Robot Manipulation",
    "authors": "Nisarga Nilavadi; Ralf Römer; Moritz Reuss; Michael Krawez; Tobias Jülg; Angela P. Schoellig; Rudolf Lioutikov; Wolfram Burgard",
    "affiliations": "Artificial Intelligence and Robotics Lab, University of Technology Nuremberg (UTN), Germany; Learning Systems and Robotics Lab, Technical University of Munich (TUM), Germany; Intuitive Robots Lab, Karlsruhe Institute of Technology (KIT), Germany; NVIDIA; Robotics Institute Germany",
    "contribution": "DUET-DINO couples side- and wrist-camera latent predictors through cross-attention, then searches seven-dimensional robot actions against paired goal images. Its clearest evidence is improved simulated spatial and orientation planning; hardware success remains limited under a smaller planning budget and safety terminations (e-architecture, e-planning, e-angled, e-hardware).",
    "abstract": "",
    "submittedDate": "2026-09-09",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv260910506",
    "arxivUrl": "https://arxiv.org/abs/2609.10506",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2609.10506",
    "pdfUrl": "https://arxiv.org/pdf/2609.10506",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "潜空间预测与JEPA",
      "三维多视角建模"
    ],
    "architecture": "待核实",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": null
  },
  {
    "id": "2609.10243",
    "title": "FolDeX: A Physical-World Benchmark for Long-Horizon Robotic Manipulation of Deformable Objects",
    "authors": "Chenhuan Liu; Yi Xu; Feng Wu; Hanyang Wang; Wenxiao Kuai; Weihao Ding; Shan Wang; Yang Liu; Shuyong Gao; Wenqiang Zhang",
    "affiliations": "Fudan University; AI Research Center, Midea Group (Shanghai) Co., Ltd.; Carnegie Mellon University",
    "contribution": "FolDeX evaluates complete physical garment-folding episodes and asks whether expensive robot experience can be reused across recovery states, tasks, scenes, and embodiments. Its reference policies use π0. Recovery-augmented training raises reported average success from 80.75% to 95.00%, but transfer studies remain preliminary and the missing quality-scoring appendix prevents independent reconstruction of FoldScore from the supplied paper alone.",
    "abstract": "",
    "submittedDate": "2026-09-09",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv260910243",
    "arxivUrl": "https://arxiv.org/abs/2609.10243",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2609.10243",
    "pdfUrl": "https://arxiv.org/pdf/2609.10243",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "机器人操作基准",
      "长时序与语言任务评测"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": null
  },
  {
    "id": "2609.10050",
    "title": "Grounding Generated Video Plans in Simulation Towards Versatile Dexterous Controllers",
    "authors": "Tianyue Wu; Boyuan An; Shuqi Zhao; Heyu Guo; Wanli Xing; Yi Ma; Kaifeng Zhang; Ruihai Wu; Masayoshi Tomizuka",
    "affiliations": "University of California, Berkeley; Sharpa Robotics; The University of Hong Kong",
    "contribution": "GALATEA turns image-and-language-conditioned manipulation videos into 3-D hand–object references, then learns simulation-based controllers that physically track them. Its central interface retains finger motion and object motion together. Multi-skill experts are distilled into one feedback policy; simulated transfer and 27/40 real-world successes on unseen plans support useful, but limited, generalization (e02, e09, e14, e17).",
    "abstract": "",
    "submittedDate": "2026-09-09",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv260910050",
    "arxivUrl": "https://arxiv.org/abs/2609.10050",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2609.10050",
    "pdfUrl": "https://arxiv.org/pdf/2609.10050",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "视觉规划与IDM",
      "灵巧操作"
    ],
    "architecture": "待核实",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": null
  },
  {
    "id": "2609.09941",
    "title": "HaWMPO: Hallucination-Aware World Model-based Policy Optimization for Generalist Robot Policy",
    "authors": "Zengjue Chen; Peidong Liu; Jiawei Li; Qi Wang",
    "affiliations": "Joy Future Academy, JD; School of Artificial Intelligence, Jilin University",
    "contribution": "HaWMPO post-trains OpenVLA-OFT inside a frozen video world model. A separate hallucination detector discounts rewards assigned to unreliable imagined action chunks before GRPO updates the policy. Table 1 reports 63.7% average LIBERO success, but gains vary by suite and the penalty-selection description is inconsistent. Physical testing provides preliminary evidence on two G1 tasks.",
    "abstract": "",
    "submittedDate": "2026-09-09",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv260909941",
    "arxivUrl": "https://arxiv.org/abs/2609.09941",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2609.09941",
    "pdfUrl": "https://arxiv.org/pdf/2609.09941",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "策略后训练与WM-RL"
    ],
    "architecture": "待核实",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": null
  },
  {
    "id": "2609.09630",
    "title": "JEPA Policy: Diffusion-Free Imitation Learning via Paired Action and Future Representation Prediction",
    "authors": "Jie Xu; Kangjin Yu; Ziyi Jin; Junjie Gao; Liqing Chen; Yixian Li; Shuai Tian; Zhongpu Xia",
    "affiliations": "Anyverse Dynamics",
    "contribution": "JEPA Policy trains a shared Transformer to predict an action chunk and the visual representation observed later in the same demonstration. Two deterministic passes refine both outputs. Its strongest evidence is improved simulated control relative to action-only MIP, supported by topology controls; future error also offers a delayed, task-dependent rollout diagnostic.",
    "abstract": "",
    "submittedDate": "2026-09-09",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv260909630",
    "arxivUrl": "https://arxiv.org/abs/2609.09630",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2609.09630",
    "pdfUrl": "https://arxiv.org/pdf/2609.09630",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "联合视频动作建模",
      "潜空间预测与JEPA"
    ],
    "architecture": "One Model",
    "predictionParadigm": "联合预测",
    "quadrant": "Q1 · One Model × 联合预测",
    "classificationStatus": null
  },
  {
    "id": "2609.09597",
    "title": "Compact Visuotactile World Models for Lifting: Prediction, Reward Alignment, and Force Constraints",
    "authors": "Qinzhen Ma; Sida Peng",
    "affiliations": "Rice University; Zhejiang University",
    "contribution": "This study asks whether better contact prediction produces better lifting decisions. A compact visuotactile model improves forecasts over vision alone, but persistence challenges the value of its dynamics. An exploratory reward revision improves executed simulator lifting, while simple force feedback remains stronger under a force budget. A separate GelSight experiment exposes the gap between frame and trajectory reliability.",
    "abstract": "",
    "submittedDate": "2026-09-09",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv260909597",
    "arxivUrl": "https://arxiv.org/abs/2609.09597",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2609.09597",
    "pdfUrl": "https://arxiv.org/pdf/2609.09597",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "多模态触觉音频",
      "策略后训练与WM-RL"
    ],
    "architecture": "待核实",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": null
  },
  {
    "id": "2609.09418",
    "title": "Valerant: An Automatic Navigable Game Map Generator via Action-Conditioned World Model Exploration",
    "authors": "Yiran Qiao; Feng Wang; Jing Ma",
    "affiliations": "Case Western Reserve University; Johns Hopkins University",
    "contribution": "Valerant expands one game screenshot into a persistent 3D point-cloud map by generating alternative action-conditioned videos, reconstructing each with SLAM, and committing the best admissible branch. A frozen video model supplies futures; an external exploration policy selects actions. Perceptual comparison and qualitative ablations support this prototype, while geometric reliability and reproducibility remain incompletely measured.",
    "abstract": "",
    "submittedDate": "2026-09-08",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv260909418",
    "arxivUrl": "https://arxiv.org/abs/2609.09418",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2609.09418",
    "pdfUrl": "https://arxiv.org/pdf/2609.09418",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "",
    "majorCategory": "Related resources",
    "subcategories": [
      "Generative-world applications"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": null
  },
  {
    "id": "2609.09155",
    "title": "SyncWorld: Visual Calibration Enables World Models as Zero-Shot Simulators",
    "authors": "Yuncong Yang; Zhengtao Han; Furkan Ozyurt; Zeyuan Yang; Han Yang; Junyi Cao; Haoyu Zhen; Yilun Du; Chuang Gan",
    "affiliations": "UMass Amherst; UC Berkeley; NYU; Harvard",
    "contribution": "SyncWorld interprets robot commands through a short visual calibration context, then predicts proposed actions’ consequences with a video diffusion model. Calibration-conditioned training and distillation also support history-only prediction. It improves held-out video metrics and selected LIBERO policy outcomes, but the evidence is strongest for short-horizon simulation: ranking quality, object interactions and unseen-embodiment controllability remain limiting factors.",
    "abstract": "",
    "submittedDate": "2026-09-08",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv260909155",
    "arxivUrl": "https://arxiv.org/abs/2609.09155",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2609.09155",
    "pdfUrl": "https://arxiv.org/pdf/2609.09155",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "视觉规划与IDM",
      "泛化与动作对齐"
    ],
    "architecture": "待核实",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": null
  },
  {
    "id": "2609.08162",
    "title": "WorldAgen: Unified State-Action Prediction with Test-Time World Model Training",
    "authors": "Chi Wan; Kangrui Wang; Yuan Si; Pingyue Zhang; Manling Li",
    "affiliations": "Northwestern University",
    "contribution": "WorldAgen shares a Transformer between action prediction and future-observation prediction, then adapts the shared representation using exploratory target-environment transitions. Simulated manipulation improves after observation-only test-time training (TTT), while implementation ambiguities and unreported uncertainty limit the strength of the generalization claim.",
    "abstract": "",
    "submittedDate": "2026-09-08",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv260908162",
    "arxivUrl": "https://arxiv.org/abs/2609.08162",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2609.08162",
    "pdfUrl": "https://arxiv.org/pdf/2609.08162",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "联合视频动作建模",
      "泛化与动作对齐"
    ],
    "architecture": "One Model",
    "predictionParadigm": "联合预测",
    "quadrant": "Q1 · One Model × 联合预测",
    "classificationStatus": null
  },
  {
    "id": "2609.07398",
    "title": "OpenWAM: An Open, Modular Exploration Towards Systematic World-Action Model Pretraining",
    "authors": "Yuran Wang; Siqiao Huang; Mingleyang Li; Chenhao Zhang; Jiaqi Liang; Weiyang Jin; Yue Chen; Xuemin Chi; Donghao Zhou; Qize Yu; Yu-Kai Wang; Yuhan Rui; Shenzhe Yao; Zhen Yuan; Zhenhao Shen; Kefei Zhu; Zijie Zhu; Ning Gao; Xiaowei Chi; Guanqi He; Shanghang Zhang; Hao Dong; Lin Shao; Hang Zhao",
    "affiliations": "National University of Singapore; Tsinghua University; Peking University; The University of Hong Kong; Zhejiang University; The Chinese University of Hong Kong; Shanghai Jiao Tong University",
    "contribution": "OpenWAM turns world–action modeling into a modular design study, then instantiates OpenWAM-α: a video DiT coupled to a dedicated ActionDiT through mutual attention and joint denoising. Its strongest lesson is conditional: embodied pretraining improves scene transfer, but strong manipulation scores coexist with substantial visual-robustness failures. The evidence below separates controlled ablations, final-model benchmarks and physical execution.",
    "abstract": "",
    "submittedDate": "2026-09-07",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv260907398",
    "arxivUrl": "https://arxiv.org/abs/2609.07398",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2609.07398",
    "pdfUrl": "https://arxiv.org/pdf/2609.07398",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "联合视频动作建模"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "联合预测",
    "quadrant": "Q3 · Dual-system × 联合预测",
    "classificationStatus": null
  },
  {
    "id": "2609.07126",
    "title": "Beyond Task Success: Stage-Wise Reliability of World Model Planning under Sensing Degradation",
    "authors": "Geonmyeong Lee; Byoung-Tak Zhang",
    "affiliations": "Seoul National University",
    "contribution": "This diagnostic study follows synthetic sensing disturbances through an existing world-model planner's representation, action-conditioned prediction, candidate ranking, and executed task outcome. Paired OGBench evaluations show that a disturbance's relative impact can change between stages. Temporal information position matters even when aggregate representation shifts are similar. These diagnostics locate sensitivity; they neither establish a general predictor of task failure nor demonstrate successful sensing mitigation.",
    "abstract": "",
    "submittedDate": "2026-09-07",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv260907126",
    "arxivUrl": "https://arxiv.org/abs/2609.07126",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2609.07126",
    "pdfUrl": "https://arxiv.org/pdf/2609.07126",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "",
    "majorCategory": "评估指标（Metrics）",
    "subcategories": [
      "行为与表征诊断",
      "统计评测协议"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": null
  },
  {
    "id": "2609.07002",
    "title": "WM-Craftnet: World Synesthesia Model for Generalizable and Robust Dexterous In-Hand Manipulation",
    "authors": "Jie Yin; Zeyuan Zhao; Xiaojing Tan; Yang Liu; Chiyu Wang; Xinyang Gu",
    "affiliations": "Sharpa Robotics",
    "contribution": "WM-Craftnet learns a predictive visuotactile state that conditions a separate dexterous manipulation policy. Its main contribution is recurrent perception for executed control: noisy depth, touch, proprioception, and action history inform a latent state trained with clean-depth and other prediction targets. Hardware rotation improves substantially, but severe-disturbance recovery remains limited (e-rssm, e-policy, e-stress).",
    "abstract": "",
    "submittedDate": "2026-09-07",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv260907002",
    "arxivUrl": "https://arxiv.org/abs/2609.07002",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2609.07002",
    "pdfUrl": "https://arxiv.org/pdf/2609.07002",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "多模态触觉音频",
      "灵巧操作"
    ],
    "architecture": "待核实",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": null
  },
  {
    "id": "2609.06578",
    "title": "Learning to Use Imagination: Progress-Conditioned Future Utilization for World Action Models",
    "authors": "Yijie Zhu; Zitong Yu; Wei Li; Hui Ma; Wen Li; Rui Shao; Liqiang Nie",
    "affiliations": "Harbin Institute of Technology (Shenzhen), China; Great Bay University, Dongguan, China; University of Electronic Science and Technology of China, Chengdu, China",
    "contribution": "ProWAM learns execution progress from recent action–observation feedback and recurrent memory, then uses that state to redistribute imagined-future attention inside a joint video–action policy. Its strongest evidence combines executed manipulation results with ablations separating progress estimation from future utilization.",
    "abstract": "",
    "submittedDate": "2026-09-06",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv260906578",
    "arxivUrl": "https://arxiv.org/abs/2609.06578",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2609.06578",
    "pdfUrl": "https://arxiv.org/pdf/2609.06578",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "联合视频动作建模",
      "记忆与长时序"
    ],
    "architecture": null,
    "predictionParadigm": null,
    "quadrant": null,
    "classificationStatus": null
  },
  {
    "id": "2609.06009",
    "title": "How to Learn from What a Human Would Avoid? Intervention-Aware World Models with Real-World RL for Dexterous Manipulation",
    "authors": "Jiaju Yin; Zhenhui Zhang; Lixin Xu; Heng Zhang; Jun Shao; Yating Feng; Arash Ajoudani; Renjing Xu",
    "affiliations": "HKUST (Guangzhou); Italian Institute of Technology; Zhejiang University",
    "contribution": "WHIRL learns when a human would take over a dexterous robot and uses that prediction to steer residual-policy training. A frozen imitation prior supplies nominal behavior; an action-conditioned, one-step latent world model supports critic learning and actor risk shaping. Five real-robot tasks show higher autonomous success, but the evidence is restricted to one training seed, one operator and familiar workspace regions.",
    "abstract": "",
    "submittedDate": "2026-09-05",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv260906009",
    "arxivUrl": "https://arxiv.org/abs/2609.06009",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2609.06009",
    "pdfUrl": "https://arxiv.org/pdf/2609.06009",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "策略后训练与WM-RL",
      "灵巧操作"
    ],
    "architecture": "待核实",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": null
  },
  {
    "id": "2609.05834",
    "title": "Learning Counterfactual World Models for Embodied Reasoning under Partial Observability",
    "authors": "Todd Y. Zhou; Daniel Zhang",
    "affiliations": "Harvard University",
    "contribution": "Counterfactual Latent World Models (CLWM) train action-conditioned imagined futures to preserve differences in intervention outcomes even when observations look alike. A recurrent world model supplies latent rollouts to model-predictive control; privileged outcome labels supervise an additional contrastive objective during training. Reported simulation gains reach 11.6 percentage points in navigation success. The evidence supports targeted representation training, while supervision-matched controls and audits of pretrained encoders remain open.",
    "abstract": "",
    "submittedDate": "2026-09-05",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv260905834",
    "arxivUrl": "https://arxiv.org/abs/2609.05834",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2609.05834",
    "pdfUrl": "https://arxiv.org/pdf/2609.05834",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "潜空间预测与JEPA",
      "策略后训练与WM-RL"
    ],
    "architecture": "待核实",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": null
  },
  {
    "id": "2609.05588",
    "title": "GE-Act 2.0: Pretraining and Scaling a World-Action Model for Robotic Manipulation",
    "authors": "AgiBot Research Team",
    "affiliations": "",
    "contribution": "GE-Act 2.0 connects a compact visual representation, a one-step future generator, and a separate inverse dynamics model. Its central training problem is pairing generated futures with compatible recorded actions. KASO selects futures using the current action model before updating both modules. Real-robot experiments show broader manipulation capability as co-training data grows, with persistent weaknesses in fine manipulation and ordinal language; simulation comparisons use a separate adaptation protocol.",
    "abstract": "",
    "submittedDate": "2026-09-04",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv260905588",
    "arxivUrl": "https://arxiv.org/abs/2609.05588",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2609.05588",
    "pdfUrl": "https://arxiv.org/pdf/2609.05588",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "视觉规划与IDM"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": null
  },
  {
    "id": "2609.05266",
    "title": "TacPAC: Tactile Prediction and Real-Time Action Correction in World-Action Models for Contact-Rich Manipulation",
    "authors": "Zipei Ma; Xiaofei Wei; Junzhe Jiang; Shunlin Lu; Li Zhang",
    "affiliations": "School of Data Science, Fudan University; Shanghai Innovation Institute; NeoteAI",
    "contribution": "TacPAC predicts future visual and tactile observations while planning an action chunk, then uses incoming touch to revise its unfinished actions against a fixed prediction-and-plan cache. On five physical manipulation tasks, it reports 64% average success versus 48% for T-Rex and 22% for its vision-only ablation. The central evidence is the controlled cache-access ablation, not visual prediction quality alone.",
    "abstract": "",
    "submittedDate": "2026-09-04",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv260905266",
    "arxivUrl": "https://arxiv.org/abs/2609.05266",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2609.05266",
    "pdfUrl": "https://arxiv.org/pdf/2609.05266",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "多模态触觉音频",
      "高效推理与实时控制"
    ],
    "architecture": null,
    "predictionParadigm": null,
    "quadrant": null,
    "classificationStatus": null
  },
  {
    "id": "2609.04911",
    "title": "TourPhysics: Bringing Physics to World Models for Exploration and Manipulation from a Single Image",
    "authors": "Xin Zhang; Yabo Chen; Zixuan Duan; Haibin Huang; Chi Zhang; Feng Xu; Xuelong Li",
    "affiliations": "Fudan University; Institute of Artificial Intelligence (TeleAI), China Telecom; Nanjing University",
    "contribution": "TourPhysics renders persistent camera tours and object manipulations from a single image plus a declared physical scene. A simulator fixes each trajectory before diffusion generates its appearance; accepted windows publish physical endpoints and visual memory together. The strongest evidence concerns prescribed motion and modest improvements on held-out revisits, with limited physical-event support and no real-world validation.",
    "abstract": "",
    "submittedDate": "2026-09-04",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv260904911",
    "arxivUrl": "https://arxiv.org/abs/2609.04911",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2609.04911",
    "pdfUrl": "https://arxiv.org/pdf/2609.04911",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "神经世界模拟器"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": null
  },
  {
    "id": "2609.04193",
    "title": "GIFT: Guided Intermediate Feature Training via Action-Oriented Structural Supervision for Robotic Manipulation",
    "authors": "Yupeng Zheng; Xiang Li; Songen Gu; Yuhang Zheng; Shuai Tian; Weize Li; Linbo Wang; Chaoyue Li; Qichao Zhang; Haoran Li; Zhongpu Xia; Ya-Qin Zhang; Shuicheng Yan; Dongbin Zhao",
    "affiliations": "Institute of Automation, Chinese Academy of Sciences; University of Chinese Academy of Sciences; Tsinghua University; Fudan University; National University of Singapore",
    "contribution": "GIFT trains robot-policy features to retain geometry, object–end-effector relations and instruction-relevant regions. The same auxiliary objectives improve three different action formulations while their default deployment uses no auxiliary predictions. Evidence is strongest for matched-baseline robustness gains, with substantial annotation requirements and uneven benefits across shifts.",
    "abstract": "",
    "submittedDate": "2026-09-03",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv260904193",
    "arxivUrl": "https://arxiv.org/abs/2609.04193",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2609.04193",
    "pdfUrl": "https://arxiv.org/pdf/2609.04193",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "泛化与动作对齐"
    ],
    "architecture": null,
    "predictionParadigm": null,
    "quadrant": null,
    "classificationStatus": null
  },
  {
    "id": "2609.03927",
    "title": "Toward Unified Robot Learning: Bridging Representation, Vision-Language-Action, and World Models",
    "authors": "Shaunak A. Mehta; Ananya Hazarika; Haochen Zhang; Fan Yang; Ryo Moriyama; Wenkai Li; Yash Patel; Kanata Suzuki",
    "affiliations": "Fujitsu Research of America; Carnegie Mellon University; Fujitsu Limited",
    "contribution": "This survey organizes robot learning around representations that encode the environment, VLA policies that generate actions, and world models that predict consequences. Its useful contribution is a vocabulary for tracing how information and feedback cross those interfaces. Integration can remain modular; the authors argue that its value should be demonstrated through improved behavior under uncertainty, distribution shift and temporal dependencies. A small original CALVIN video-prediction diagnostic illustrates hidden-object and collision failures, but supplies no quantitative proof that a unified architecture resolves them.",
    "abstract": "",
    "submittedDate": "2026-09-03",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv260903927",
    "arxivUrl": "https://arxiv.org/abs/2609.03927",
    "codeUrls": [],
    "projectUrl": null,
    "venue": "Transactions on Machine Learning Research (2026)",
    "paperUrl": "https://arxiv.org/abs/2609.03927",
    "pdfUrl": "https://arxiv.org/pdf/2609.03927",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "",
    "majorCategory": "Related resources",
    "subcategories": [
      "Surveys & perspectives"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": null
  },
  {
    "id": "2609.03681",
    "title": "WISE: World-model-guided Imagination Scheduling for Efficient Post-training of Vision-Language-Action Models",
    "authors": "Chenhao Zhang; Hanyu Zhao; Hang Cheng; Tengfei Pan; Long Zeng",
    "affiliations": "Tsinghua University; Beijing Academy of Artificial Intelligence (BAAI)",
    "contribution": "WISE post-trains a VLA action head by selectively imagining alternative behaviors at visually identified interaction states. A separate frozen world model predicts bounded futures; a frozen evaluator ranks them; updates supervise only the first action chunk from a real context. The strongest controlled evidence is the scheduling ablation: higher task success with substantially less imagination computation. The method, results, and reproduction boundaries below trace this conclusion to the primary text.",
    "abstract": "",
    "submittedDate": "2026-09-03",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv260903681",
    "arxivUrl": "https://arxiv.org/abs/2609.03681",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2609.03681",
    "pdfUrl": "https://arxiv.org/pdf/2609.03681",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "策略后训练与WM-RL"
    ],
    "architecture": "待核实",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": null
  },
  {
    "id": "2609.03602",
    "title": "SV-WAM: An Efficient Surround-View World-Action Model for End-to-End Autonomous Driving",
    "authors": "Jinyang Wang; Shiwei Li; Junjian Wang; Zhiqiang Deng; Jianbin Gao; Yihang Zhao; Liu Liu; Yongjia Zhao; Jinlong Chen; Huirui Xu; Yifeng Pan; Kangwei Liu; Fan Ren; Ji Tao; Minghao Yang",
    "affiliations": "Institute of Automation, Chinese Academy of Sciences; Chongqing Changan Technology Co., Ltd.; Civil Aviation University of China; Beihang University; Guilin University of Electronic Technology",
    "contribution": "SV-WAM trains one shared transformer to denoise driving actions and future surround-view video, but blocks action attention to future-video tokens. Deployment retains six-camera history while generating only trajectories. A differentiable vehicle-footprint loss improves road compliance. The strongest reported result is 91.0 EPDMS on NAVSIMv2 navtest; the evidence concerns benchmark planning, with weaker hard-split performance and no demonstrated real-vehicle deployment.",
    "abstract": "",
    "submittedDate": "2026-09-03",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv260903602",
    "arxivUrl": "https://arxiv.org/abs/2609.03602",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2609.03602",
    "pdfUrl": "https://arxiv.org/pdf/2609.03602",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "自动驾驶",
      "联合视频动作建模"
    ],
    "architecture": "One Model",
    "predictionParadigm": "联合预测",
    "quadrant": "Q1 · One Model × 联合预测",
    "classificationStatus": null
  },
  {
    "id": "2609.03572",
    "title": "Drive-HWM: Hierarchical World Models for Dynamic-Latent Guided Autonomous Driving",
    "authors": "Zhaoxin Fan; Tianbao Zhang; Wenjun Wu; Xiaofeng Wang; Yeying Jin; Jian Zhao; Zheng Zhu; Shuicheng Yan",
    "affiliations": "School of Artificial Intelligence, Beihang University, Beijing, China; Shanghai Jiao Tong University, Shanghai, China; Dim12 AI; GigaAI; National University of Singapore, Singapore; TeleAI",
    "contribution": "Drive-HWM couples a periodically refreshed predictor of future motion latents with an observation-grounded autoregressive driving policy. Optical flow supervises the slow branch; next-frame RGB tokens supervise the fast branch during training. NAVSIM tables report stronger aggregate driving scores, but conflicting prose and incomplete implementation details limit precise reproduction.",
    "abstract": "",
    "submittedDate": "2026-09-03",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv260903572",
    "arxivUrl": "https://arxiv.org/abs/2609.03572",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2609.03572",
    "pdfUrl": "https://arxiv.org/pdf/2609.03572",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "自动驾驶",
      "潜空间预测与JEPA"
    ],
    "architecture": null,
    "predictionParadigm": null,
    "quadrant": null,
    "classificationStatus": null
  },
  {
    "id": "2609.03565",
    "title": "Toward Physically Grounded JEPA World Models for Goal-Conditioned Robotic Planning",
    "authors": "Muyuan Liu; Yue Huang; Zheng Liang; Xiang Gao",
    "affiliations": "GENISOM AI, Beijing, China",
    "contribution": "SA+IDM trains an action-conditioned JEPA world model with two auxiliary heads: inverse dynamics recovers executed actions, while state alignment predicts measured physical state from consecutive image representations. Deployment uses only the encoder and latent predictor inside CEM planning. State alignment improves all four reported tasks over IDM alone, while the diagnostics challenge average temporal straightening as a sufficient representation-quality criterion (e2–e12).",
    "abstract": "",
    "submittedDate": "2026-09-03",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv260903565",
    "arxivUrl": "https://arxiv.org/abs/2609.03565",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2609.03565",
    "pdfUrl": "https://arxiv.org/pdf/2609.03565",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "潜空间预测与JEPA"
    ],
    "architecture": "待核实",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": null
  },
  {
    "id": "2609.03557",
    "title": "Building Pretraining Data for World Models: An Unreal Engine-Based Pipeline for Action-Conditioned Video Generation",
    "authors": "Haoyu Wang; Songchun Zhang; Haoran Li; Haoyang Huang; Zeyue Xue; Nan Duan",
    "affiliations": "Joy Future Academy, JD; Tsinghua University; The Hong Kong University of Science and Technology",
    "contribution": "This paper documents the Unreal Engine synthetic-data component used in EchoWM: simulate character motion once, record controls and states, then replay the trajectory for high-quality multi-view rendering. Its contribution is a production system combining scene curation, cache locality and failure recovery. Reported video volume establishes production scale; downstream learning utility remains untested here. [e-scope, e-workflow, e-scale, e-limits]",
    "abstract": "",
    "submittedDate": "2026-09-03",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv260903557",
    "arxivUrl": "https://arxiv.org/abs/2609.03557",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2609.03557",
    "pdfUrl": "https://arxiv.org/pdf/2609.03557",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "",
    "majorCategory": "数据集",
    "subcategories": [
      "合成数据与数据生成",
      "数据采集接口"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": null
  },
  {
    "id": "2609.03294",
    "title": "Latent Energy Action Planning with World Models",
    "authors": "Phu Pham and Aniket Bera",
    "affiliations": "Department of Computer Science, Purdue University, USA",
    "contribution": "LEAP refines an action horizon through frozen LeWorldModel dynamics, combining latent-goal matching with a learned decoder’s terminal-state error. A trained proposal initializes search; projection bounds executed controls. Four-domain mean success rises from 77.5% to 94.8% against matched LeWM+CEM. The narrower energy ablation improves from 91.0% to 96.5%, separating the extra objective’s contribution from the complete planner change. Numerical goal descriptors and trustworthy learned rollouts remain prerequisites (e-energy, e-proposal, e-optimization, e-main, e-ablation, e-limits).",
    "abstract": "",
    "submittedDate": "2026-09-03",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv260903294",
    "arxivUrl": "https://arxiv.org/abs/2609.03294",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2609.03294",
    "pdfUrl": "https://arxiv.org/pdf/2609.03294",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "潜空间预测与JEPA"
    ],
    "architecture": "待核实",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": null
  },
  {
    "id": "2609.02531",
    "title": "Spatially Aware World Action Model via Geometric Latent Diffusion",
    "authors": "Gonzalez, Javier Alejandro Lopetegui; Pacaud, Paul; Schmid, Cordelia",
    "affiliations": "de l'École Normale Supérieure, PSL Research University in Paris",
    "contribution": "To bridge this gap, we introduce a Spatially Aware World Action Model (SA-WAM), which repurposes a pretrained video model for joint action, RGB, and depth prediction, enabling 3D-aware world modeling and action prediction within a single diffusion backbone. SA-WAM achieves state-of-the-art results on the RoboCasa and LIBERO-Plus benchmarks, while simultaneously improving future-state predictions.",
    "abstract": "World Action Models (WAMs) leverage the capabilities of large-scale pretrained video diffusion models to jointly predict future observations and actions, inheriting rich visual and physical priors from internet-scale video. This has made them a promising paradigm for robot policy learning, yet the prevailing models operate exclusively on RGB observations and do not leverage 3D information. To bridge this gap, we introduce a Spatially Aware World Action Model (SA-WAM), which repurposes a pretrained video model for joint action, RGB, and depth prediction, enabling 3D-aware world modeling and action prediction within a single diffusion backbone. We use a nonlinear encoding that maps the unbounded depth signal into the bounded input domain expected by the frozen VAE tokenizer. This allows us to reuse the tokenizer without 3D-specific fine-tuning, incorporating geometric information without sacrificing the pretrained priors. SA-WAM achieves state-of-the-art results on the RoboCasa and LIBERO-Plus benchmarks, while simultaneously improving future-state predictions. Furthermore, SA-WAM outperforms strong baselines in real-world evaluation using a UR5 robotic arm, with strong gains in randomized environments. We analyze the correlation between world model prediction quality and rollout success, providing insights into WAM performance and avenues for its improvement.",
    "submittedDate": "2026-09-02",
    "primaryCategory": "3D/4D WAM",
    "secondaryCategories": [
      "Evaluation / Survey / Theory",
      "Latent / Representation WAM"
    ],
    "bibtexKey": "SAWAM",
    "arxivUrl": "https://arxiv.org/abs/2609.02531",
    "codeUrls": [],
    "projectUrl": "https://jlopetegui98.github.io/projects/sa_wam.html",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2609.02531",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "联合视频动作建模",
      "三维多视角建模"
    ],
    "architecture": "One Model",
    "predictionParadigm": "联合预测",
    "quadrant": "Q1 · One Model × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2609.02159",
    "title": "World-Coherent Decoding: Self-Verifying Test-Time Planning for World Action Models",
    "authors": "Zhang, Chuhan; Ito, Seiji; Hoshino, Kenta; Ikehata, Satoshi; Sato, Ikuro",
    "affiliations": "Department of Computer Science; Institute of Science Tokyo; DENSO IT Lab, Japan; National Institute of Informatics, Japan",
    "contribution": "We propose World-Coherent-Decoding (WCD), a self-verifying test-time planning framework that treats WAM rollouts as falsifiable future--action hypotheses.",
    "abstract": "World Action Models (WAMs) aim to control robots by stochastically generating visual futures and then decoding actions, but empirical observations indicate that the results can strongly depend on which future is selected. We propose World-Coherent-Decoding (WCD), a self-verifying test-time planning framework that treats WAM rollouts as falsifiable future--action hypotheses. At each decision step, WCD samples multiple candidates from a frozen WAM and ranks them using internal generative signals: flow-based video surprisal for visual plausibility and action path effort for action-generation stability. After execution, the realized observation audits the selected imagination, yielding an imagination--reality mismatch that trains a lightweight online predictor for future candidate selection. Thus, WCD converts delayed self-verification into pre-execution reliability estimation without updating the backbone model. On RoboTwin 2.0, WCD improves Hard success under limited randomized-scene supervision from 55.80%55.80\\% to 60.90%60.90\\%, with a +16.43+16.43 gains on Horizon-3 tasks, and shows qualitative robustness on real Franka visual-shift tests. These results highlight a simple principle: test-time scaling for WAMs depends less on sampling more futures than on selecting reliable ones.",
    "submittedDate": "2026-09-02",
    "primaryCategory": "Real-Time / Efficient WAM",
    "secondaryCategories": [],
    "bibtexKey": "WorldCoherentDecoding",
    "arxivUrl": "https://arxiv.org/abs/2609.02159",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2609.02159",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "策略后训练与WM-RL",
      "高效推理与实时控制"
    ],
    "architecture": "不适用",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.24882",
    "title": "Latent Action as Intention Enables Efficient Future Imagination for World Action Models",
    "authors": "Li, Xiang; Zheng, Yupeng; Gu, Songen; Ma, Huailiang; Yu, Feng; Zheng, Yuhang; Nie, Xian; Yuan, Shanshuai; Zang, Yujie; Li, Weize; Tian, Shuai; Liu, Moyang; Zhang, Ya-Qin; Ding, Wenchao",
    "affiliations": "Not identified",
    "contribution": "To bridge this gap, we introduce LAWA, a WAM architecture that uses compact latent actions as an operational representation of future intentions, enabling efficient test-time future imagination without generating future observations. On RoboCasa, LAWA achieves state-of-the-art average success rates of 65.6% and 80.8% in the few-shot and full data settings, improving over the matched Fast-WAM baseline by 9.6 and 4.5 points, respectively.",
    "abstract": "World action models (WAMs) improve robot control by modeling how observations evolve, but generating future observations at test time incurs substantial latency. Fast-WAM removes this process for efficiency; however, our matched implementations show lower generalization for Fast-WAM than for future-aware alternatives, especially with scarce robot demonstrations and in out-of-distribution scenarios. To bridge this gap, we introduce LAWA, a WAM architecture that uses compact latent actions as an operational representation of future intentions, enabling efficient test-time future imagination without generating future observations. Specifically, a discrete tokenizer enhanced by action-free pre-training produces manipulation-centric codebook targets. LAWA jointly denoises a continuous latent state anchored to these targets with executable action chunks while omitting the future-video branch at inference. On RoboCasa, LAWA achieves state-of-the-art average success rates of 65.6% and 80.8% in the few-shot and full data settings, improving over the matched Fast-WAM baseline by 9.6 and 4.5 points, respectively. It also preserves the performance level of the matched Joint-WAM variant while requiring 42.9% lower inference latency. LAWA also demonstrates competitive zero-shot robustness on LIBERO-Plus and superior performance on real-world tasks. These results show that future imagination need not be discarded: retaining it with compact latent actions yields an effective trade-off among performance, generalization, and latency. Code and models will be released.",
    "submittedDate": "2026-09-01",
    "primaryCategory": "Latent / Representation WAM",
    "secondaryCategories": [
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "LAIEEFI",
    "arxivUrl": "https://arxiv.org/abs/2608.24882",
    "codeUrls": [],
    "projectUrl": "https://getterupper.github.io/LAWA",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.24882",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "潜空间预测与JEPA",
      "联合视频动作建模",
      "高效推理与实时控制"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "联合预测",
    "quadrant": "Q3 · Dual-system × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2609.00188",
    "title": "ZimaBlue: Evolving Generalizable World Action Models through Scalable Video Pre-training",
    "authors": "Wu, Xionghao; Yang, Yijun; Zhou, Shiyang; Sun, Haoze; Liu, Jianhui; Yu, Songsong; Zhang, Jiyao; Li, Wenbo; Wang, Bo; Ma, Guoqing; Song, Lin; Liao, Renjie; Zheng, Shenghe; Tang, Wei; Qi, Xiaojuan; Li, Yanwei; Zhang, Yuan; Tian, Zhuotao; Huang, Haoyang; Duan, Nan",
    "affiliations": "Joy Future Academy 1",
    "contribution": "We introduce ZimaBlue, a scalable framework for learning generalizable World Action Models (WAMs) from large-scale video.",
    "abstract": "Robotic manipulation faces a fundamental scaling challenge: robust generalization demands broad physical experience, yet action-labeled robot trajectories are expensive to collect and inherently limited in diversity. Egocentric videos offer a far more scalable source of embodied experience, capturing object interactions, contact dynamics, tool use, and long-horizon behaviors across diverse environments. The central challenge is how to convert this abundant but action-free experience into effective robot control. We introduce ZimaBlue, a scalable framework for learning generalizable World Action Models (WAMs) from large-scale video. ZimaBlue follows a three-stage training curriculum: it first performs causal embodied video pre-training on large-scale human and robot egocentric videos, then grounds the learned visual dynamics in heterogeneous robot trajectories through video-action mid-training with a unified action representation, and finally specializes the model to a target robot for deployment. To make generative WAMs practical for real-time control, ZimaBluefurther adopts an asynchronous Slow-Fast dual-system architecture, where a high-capacity Slow world model provides generalizable spatiotemporal representations and a lightweight Fast branch enables 30 Hz action prediction on NVIDIA RTX 4090. On real-robot zero-shot evaluations, scaling from target-robot data alone to over 120,000 hours of embodied video improves success from 36.1% to 77.8%. ZimaBlue further delivers strong performance across multiple benchmarks, with particularly pronounced gains on unseen tasks.",
    "submittedDate": "2026-08-31",
    "primaryCategory": "Real-Time / Efficient WAM",
    "secondaryCategories": [],
    "bibtexKey": "ZimaBlue",
    "arxivUrl": "https://arxiv.org/abs/2609.00188",
    "codeUrls": [],
    "projectUrl": "https://zimablue-wam.github.io/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2609.00188",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "视觉规划与IDM",
      "高效推理与实时控制",
      "泛化与动作对齐"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.30237",
    "title": "Motus2: A Self-Evolving General World Model for Dexterous Manipulation",
    "authors": "Bi, Hongzhe; Zhou, Zihao; Tang, Yihang; Pang, Jingrui; Huang, Shuhe; Liu, Haitian; Wang, Runqing; Huang, Shuai; Wang, Yichen; Cheng, Yiming; Zhao, Ruowen; Li, Zhenghua; Tan, Hengkai; Liu, Xiaolong; Wan, Jinhui; Liu, Jiabao; Zhao, Min; Bao, Fan; Zhu, Jun",
    "affiliations": "2 Tsinghua University",
    "contribution": "We present Motus2, a self-evolving general world model for dexterous manipulation.",
    "abstract": "General embodied agents should perceive, predict, act, evaluate, and improve within a unified system. World models have shown great promise in building such agents, yet existing models typically append an action output head to a world simulator, without coupling them into a closed decision-and-learning loop for policy improvement. We present Motus2, a self-evolving general world model for dexterous manipulation. Motus2 advances world modeling through model scaling and data scaling. For model scaling, a single model with shared weights exposes three control interfaces: a policy (world-action model), a simulator (action-conditioned world model), and an evaluator (value model). The policy proposes candidate action chunks, the simulator predicts their visual consequences, and the evaluator assesses the predicted outcomes. Their coupling forms a closed decision-and-learning loop for policy improvement. This formulation uses curated expert demonstrations for action learning, while failed and suboptimal interactions provide valuable evidence for dynamics modeling and value learning. For data scaling, Motus2 progresses from large-scale monocular egocentric data to synchronized stereo egocentric data, followed by robot-domain adaptation with robot trajectories and supplementary human-robot alignment data. Motus2 further studies global-autoregressive and hybrid-memory extensions of its sliding-window context, adds tactile feedback for contact-aware control, and is instantiated on a fully biomimetic platform with stereo vision, dual arms, dual dexterous hands, and tactile sensing. Together, egocentric data scaling and closed-loop general world model scaling provide a general path toward self-evolving dexterous manipulation.",
    "submittedDate": "2026-08-31",
    "primaryCategory": "WAM + RL",
    "secondaryCategories": [
      "Memory WAM",
      "Multimodal / Tactile WAM"
    ],
    "bibtexKey": "Motus2",
    "arxivUrl": "https://arxiv.org/abs/2608.30237",
    "codeUrls": [],
    "projectUrl": "https://motus-robotics.github.io/motus2/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.30237",
    "pdfUrl": "https://arxiv.org/pdf/2608.30237",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{Motus2,\n  title={Motus2: A Self-Evolving General World Model for Dexterous Manipulation},\n  author={Bi, Hongzhe and Zhou, Zihao and Tang, Yihang and Pang, Jingrui and Huang, Shuhe and Liu, Haitian and Wang, Runqing and Huang, Shuai and Wang, Yichen and Cheng, Yiming and Zhao, Ruowen and Li, Zhenghua and Tan, Hengkai and Liu, Xiaolong and Wan, Jinhui and Liu, Jiabao and Zhao, Min and Bao, Fan and Zhu, Jun},\n  journal={arXiv:2608.30237},\n  year={2026}\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "策略后训练与WM-RL",
      "记忆与长时序",
      "多模态触觉音频"
    ],
    "architecture": "One Model",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.22067",
    "title": "DELE-w0.5: Inferring Action from Future Latent State for Robotic Manipulation",
    "authors": "Lei, Fenghao; Huang, Zhixiong; Yang, Long; Chen, Jiabao; Huang, Peilin; Fu, Han; Li, Zhuo; Ren, Xiaoxue",
    "affiliations": "DeepLeap Technology Co., Ltd., Shenzhen, China",
    "contribution": "In this paper, we propose DELE-w0.5, which infers robot actions from predicted future states without relying on video generation. Across 640 real-robot trials on four long-horizon manipulation tasks, our DELE-w0.5 achieves the best performance among all compared policies, attaining 62.5% overall full-task success and 81.3% macro ordered-stage progress.",
    "abstract": "World-Action Models (WAMs) build robot control on video-generation backbones, which jointly predict dense future visual trajectories and robot actions. We argue that video generation is an unnecessary intermediate objective for world-action modeling. For robotic manipulation, the goal of a world model is not to reproduce how the world looks at every intermediate moment, but to predict the state that the world will reach after an action is executed. The intermediate frames only describe the visual transition between physical states, which consumes substantial model capacity and computation, but do not directly specify the physical outcome that the robot action is intended to produce. In this paper, we propose DELE-w0.5, which infers robot actions from predicted future states without relying on video generation. Concretely, DELE-w0.5 infers the action sequence from its corresponding compact future latent state. The future latent state captures the action-relevant physical outcome of robot interaction and serves as an explicit bridge between world modeling and action generation. The core design principle of DELE-w0.5 is to model how the physical world changes under robot actions, rather than how its visual appearance evolves frame by frame. This formulation removes the high-dimensional visual redundancy introduced by dense video representations, and it therefore enables cheaper training and low-latency inference. Across 640 real-robot trials on four long-horizon manipulation tasks, our DELE-w0.5 achieves the best performance among all compared policies, attaining 62.5% overall full-task success and 81.3% macro ordered-stage progress. It outperforms the strongest baseline by 32.5 percentage points in full-task success and 20.1 percentage points in macro progress.",
    "submittedDate": "2026-08-31",
    "primaryCategory": "Latent / Representation WAM",
    "secondaryCategories": [
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "DELEw05",
    "arxivUrl": "https://arxiv.org/abs/2608.22067",
    "codeUrls": [],
    "projectUrl": "https://deepleap-x.com/research/dele-w0.5",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.22067",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "潜空间预测与JEPA",
      "联合视频动作建模",
      "高效推理与实时控制"
    ],
    "architecture": "One Model",
    "predictionParadigm": "联合预测",
    "quadrant": "Q1 · One Model × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.29937",
    "title": "AcrossWAM1.0:A Modular Latent World-Action Stack for Compact Robot Policies",
    "authors": "Zhang, Yafei; Wu, Nan",
    "affiliations": "2 Institute of Automation, Chinese Academy of Sciences, Beijing, China",
    "contribution": "We introduce AcrossWAM1.0, a modularization and scaling study of this latent world-action stack.",
    "abstract": "Latent world-action models avoid rendering future pixels by predicting an action-relevant visual subgoal in feature space. LaWAM established this formulation, but its original presentation left the world model, multimodal backbone, and deployment checkpoint tightly coupled. We introduce AcrossWAM1.0, a modularization and scaling study of this latent world-action stack. Rather than presenting latent subgoals as a new algorithm, we make the module boundary explicit: a policy adapter produces latent-action and action-generation contexts; a retained latent world decoder grounds the predicted transition in the current scene;and a flow-matching expert generates continuous action chunks. We further separate training-only teachers from the inference graph and provide a verifiable deployment export. On 2,000 paired LIBERO episodes, replacing a Qwen3-VL-2B backbone with Qwen3.5-0.8B yields 97.45% success versus 98.00% for the 2B model (a-0.55percentage-point difference; exact McNemarp=0.266). This does not prove equivalence, but it meets a prespecified two-point retention criterion. The compact, inference-reachable checkpoint contains 1,472.6M unique parameters, 42.4% fewer than the original 2B policy, while all retained tensors are bitwise identical to the source checkpoint. Cross-family execution is additionally checked with a MiniCPM-V adapter smoke test; closed-loop cross-family transfer remains an open evaluation. AcrossWAM1.0 therefore contributes an auditable software and evaluation boundary for compact latent world-action policies, distinct from LaWAM's original latent-subgoal contribution.",
    "submittedDate": "2026-08-30",
    "primaryCategory": "Latent / Representation WAM",
    "secondaryCategories": [
      "Evaluation / Survey / Theory",
      "Multimodal / Tactile WAM",
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "AcrossWAM10",
    "arxivUrl": "https://arxiv.org/abs/2608.29937",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.29937",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "潜空间预测与JEPA",
      "视觉规划与IDM",
      "高效推理与实时控制"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.10860",
    "title": "Flex-ππ: A Multi-Stream World-Action Model with Compute Flexibility",
    "authors": "Yan, Ge; Liu, Jinghao; Fan, Yuzhi; Cai, Lei; Liao, Minwen; Zhang, Jesse; Fox, Dieter",
    "affiliations": "University of Washington; Allen Institute for AI",
    "contribution": "World-action models (WAMs) predict the future to act better, but nearly all of them predict only RGB latents, trained purely for pixel reconstruction, with no explicit signal for the 3D geometry or object semantics manipulation needs. We find a surprising free lunch: the same frozen video-generation VAE that encodes RGB also encodes 3D pointmaps almost losslessly, with no pointmap-specific training at all.",
    "abstract": "World-action models (WAMs) predict the future to act better, but nearly all of them predict only RGB latents, trained purely for pixel reconstruction, with no explicit signal for the 3D geometry or object semantics manipulation needs. We find a surprising free lunch: the same frozen video-generation VAE that encodes RGB also encodes 3D pointmaps almost losslessly, with no pointmap-specific training at all. This lets us supervise Flex-ππ, a 6B-parameter WAM, on 3D geometry and object-centric DINO semantics alongside RGB, at no cost in new sensors, new pre-training, or inference latency. Every visual signal is projected into this shared latent space and denoised jointly with actions inside a Mixture-of-Transformers backbone; per-stream dropout with cross-modality forcing then lets a single trained checkpoint run on any subset of these streams, from a fast action-only mode to full joint generation. The result is a policy that is exceptionally demonstration-efficient and generalizes well, beating the strongest baselines by up to 2-7×\\times on dexterous, precise, real-world bimanual manipulation tasks both in and out of distribution, all while running faster than π0.5π_{0.5}. Our project website: https://flex-pi.github.io/",
    "submittedDate": "2026-08-29",
    "primaryCategory": "Real-Time / Efficient WAM",
    "secondaryCategories": [
      "3D/4D WAM",
      "Latent / Representation WAM"
    ],
    "bibtexKey": "FlexPiPi",
    "arxivUrl": "https://arxiv.org/abs/2608.10860",
    "codeUrls": [
      "https://github.com/geyan21/flex-pi"
    ],
    "projectUrl": "https://flex-pi.github.io/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.10860",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "联合视频动作建模",
      "三维多视角建模",
      "泛化与动作对齐"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "联合预测",
    "quadrant": "Q3 · Dual-system × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.28995",
    "title": "Hydra: A Navigation World Action Model with Discrete Latent Planning and Continuous Flow-Matching Execution",
    "authors": "Nazeri, Mohammad; Card, Alexandyr; Huber, Samira; Pokhrel, Anuj; Wang, Yujun; Hammele, Ruben; Song, Daeun; Pirk, Sören; Xiao, Xuesu",
    "affiliations": "Mohammad Nazeri George Mason University, mnazerir, acard, apokhrel, xiao @gmu.edu; Samira Huber Kiel University, samira.huber, ruben.hammele, sp @informatik.uni-kiel.de; Yujun Wang Ludwig Maximilian University Munich, yujun wang; Daeun Song Ewha Womans University,",
    "contribution": "In this paper, we present Hydra, a discrete World Action Model that closes this gap by moving the planner, both the sampler and the evaluator, inside the model. Evaluated on two physical robotic platforms, Hydra outperforms state-of-the-art world models in goal-directed planning, while matching or exceeding the closed-loop execution capabilities of leading reactive foundation policies.",
    "abstract": "World models let robots imagine possible futures, but exploiting this capability for real-time control is bottlenecked by a representation misalignment: the generative model and the planner operate on decoupled manifolds, so the planner has no shared structure to search over and must instead decode every candidate back into high-dimensional pixel space to evaluate it. This decoding step is a major obstacle to real-time control on physical hardware. In this paper, we present Hydra, a discrete World Action Model that closes this gap by moving the planner, both the sampler and the evaluator, inside the model. Hydra establishes a unified latent manifold over visual states, physical poses, and control actions, then compresses this manifold through modality-specific Vector-Quantized bottlenecks into discrete vocabularies of kinodynamic intents and visual states. Because candidates are now drawn directly from this shared manifold, sampling is informed by the model's own understanding of the observation rather than proposed blind, and evaluation happens natively within the discrete space: candidates are ranked by a Kinematic-Perceptual Cost, without ever decoding to pixels. We term this Discrete Latent Planning (DLP). Because planning over discrete intents alone cannot supply the smooth, continuous commands physical actuation requires, Hydra pairs DLP with conditional Flow Matching, which maps each selected intent to a continuous trajectory for execution. Evaluated on two physical robotic platforms, Hydra outperforms state-of-the-art world models in goal-directed planning, while matching or exceeding the closed-loop execution capabilities of leading reactive foundation policies.",
    "submittedDate": "2026-08-28",
    "primaryCategory": "Latent / Representation WAM",
    "secondaryCategories": [
      "Evaluation / Survey / Theory",
      "Navigation / Driving / Domain WAM",
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "Hydra",
    "arxivUrl": "https://arxiv.org/abs/2608.28995",
    "codeUrls": [],
    "projectUrl": "https://robotixx.github.io/hydra",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.28995",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "导航",
      "潜空间预测与JEPA"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.27550",
    "title": "Beyond Data Scaling: Representation-Centric Continued Pre-training for Vision-Language-Action Models",
    "authors": "Yang, Senqiao; Wang, Chengyao; Chen, Yuxin; Wang, Zixuan; Tang, Longxiang; Gui, Haokun; Ye, Jinhui; Lu, Changsheng; Wu, Xiaoyang; Zhu, Mingkang; Chen, Pengguang; Liu, Shu; Tian, Zhuotao; Zhao, Hengshuang; Yu, Bei; Jia, Jiaya",
    "affiliations": "Not identified",
    "contribution": "We propose VLAct, a VLA-oriented VLM backbone trained on broad, heterogeneous, multi-embodiment robot data before task-specific fine-tuning. On RoboDojo, VLAct ranks sixth among all policies by success rate and outperforms all explicitly designated world-action model (WAM) entries on both metrics.",
    "abstract": "Scaling robot data is crucial for building generalist Vision-Language-Action (VLA) models, yet robot trajectories are harder to scale than web-scale image-text data because embodied collection is costly and sparsely covers the physical world. This makes representation quality a central bottleneck: under a fixed robot-data budget, continued pre-training must turn limited trajectories into transferable visual-action knowledge rather than merely fit actions. We propose VLAct, a VLA-oriented VLM backbone trained on broad, heterogeneous, multi-embodiment robot data before task-specific fine-tuning. VLAct preserves the broad VLM prior and encourages shared action semantics across embodiments through VLM-prior preservation, multi-head continuous action co-supervision, and a partially unified cross-embodiment action layout, while allowing task-specific action heads during fine-tuning. Across simulation, real-world, and unseen-embodiment transfer, VLAct consistently improves downstream performance under fixed fine-tuning protocols. On LIBERO-Plus and RoboTwin 2.0, VLAct surpasses industrial VLA systems including ABot-M0 and LingBot-VLA, achieving success rates of 82.6% and 92.5%. On RoboDojo, VLAct ranks sixth among all policies by success rate and outperforms all explicitly designated world-action model (WAM) entries on both metrics. Most notably, on RoboCasa-GR1, an unseen humanoid embodiment, VLAct using only 20% of downstream trajectories outperforms the full-data GR00T-N1.6 baseline. These results are obtained using fully open-source data and only a 16-GPU training setup, showing that representation-centric continued pre-training can deliver highly competitive performance under a modest compute budget and is an important independent axis of VLA progress beyond data scaling.",
    "submittedDate": "2026-08-27",
    "primaryCategory": "Latent / Representation WAM",
    "secondaryCategories": [
      "Navigation / Driving / Domain WAM"
    ],
    "bibtexKey": "VLAct",
    "arxivUrl": "https://arxiv.org/abs/2608.27550",
    "codeUrls": [
      "https://github.com/starVLA/VLAct"
    ],
    "projectUrl": "https://starvla.github.io/VLAct/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.27550",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "VLA",
    "subcategories": [
      "VLA后训练与数据增强"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.27259",
    "title": "Making Latent Evolution Explicit: Operator-Structured Transitions for World Action Models",
    "authors": "Lu, Xiaoxiao; Dong, Yunlong; Shi, Jiahao; Yuan, Ye",
    "affiliations": "1School of Artificial Intelligence and Automation, Huazhong University of Science and Technology 2Principia AI",
    "contribution": "We introduce the Latent Evolution Operator Network (LEON), which models latent evolution in a learned observable space through context-modulated operator-based propagation and additive forcing.",
    "abstract": "World Action Models (WAMs) augment robot policies by predicting how task-relevant scene states may evolve under interaction. Recent WAMs increasingly perform such prediction in latent representation spaces, avoiding full appearance-level generation while preserving control-relevant information. Yet latent transitions are commonly realized with Transformer-based predictors whose inductive structure is centered on token interaction rather than temporal evolution. We study transition realization as an architectural choice distinct from predictive representation and prediction-policy coupling. We introduce the Latent Evolution Operator Network (LEON), which models latent evolution in a learned observable space through context-modulated operator-based propagation and additive forcing. Grounded in the controlled Koopman generator view of evolution, LEON organizes context-dependent transition variation around a shared evolution-operator structure while retaining a complementary path for additive change. Controlled dynamical systems verify the resulting evolution-specific inductive bias and the complementary roles of operator propagation and forcing. Across two WAM formulations that integrate latent prediction into the policy differently, LEON improves closed-loop performance and robustness while remaining effective under full transition replacement. These results establish transition realization as a consequential architectural choice in latent WAMs.",
    "submittedDate": "2026-08-27",
    "primaryCategory": "Latent / Representation WAM",
    "secondaryCategories": [],
    "bibtexKey": "LEON",
    "arxivUrl": "https://arxiv.org/abs/2608.27259",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.27259",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "潜空间预测与JEPA",
      "泛化与动作对齐"
    ],
    "architecture": "不适用",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.27033",
    "title": "Riemann-1.0: An Embodied World Action Model for Physical AI",
    "authors": "Sun, Haofeng; Pei, Jiangbo; Kang, Fei; Liu, Zexiang; Li, Yaokun; Jiang, Boyi; Xue, Hua; Zhou, Cindy; Li, Wei; Wei, Yichen; An, Mengyin; Zhao, Fanliang; Jiang, Biao; Wang, Zile; Liu, Yang; Li, Yangguang",
    "affiliations": "Not identified",
    "contribution": "We introduce Riemann-1.0, a fully causal autoregressive World Action Model for embodied intelligence. Riemann-1.0 achieves state-of-the-art performance across both simulation benchmarks and real-world manipulation tasks.",
    "abstract": "We introduce Riemann-1.0, a fully causal autoregressive World Action Model for embodied intelligence. Riemann-1.0 jointly models multi-view visual observations, robot states, and embodiment-specific actions within a unified causal autoregressive sequence, representing robot actions and world evolution as causal state transitions. Unlike existing WAMs based on joint generation, video-first prediction, or decoupled modeling paradigms, Riemann-1.0 unifies online robot policy execution and action-conditioned world simulation within a single model, enabling it to function as both an executable robot policy and a multi-embodiment visual world simulator. To scale embodied experience across heterogeneous data sources, we further develop a progressive embodied pretraining framework that unifies learning from egocentric human videos, handheld-gripper demonstrations, and heterogeneous robot trajectories under a shared World Action Modeling objective. Built upon 200K+ hours of interaction data, Riemann-1.0 progressively transfers large-scale embodied experience into executable robot manipulation capabilities. Riemann-1.0 achieves state-of-the-art performance across both simulation benchmarks and real-world manipulation tasks. It achieves success rates of 94.3% on RoboTwin2.0, 99.0% on LIBERO, and 62.6% on the long-horizon compositional benchmark RoboCasa-365, outperforming the previous best method by 8.4% On long-horizon real-world manipulation tasks, Riemann-1.0 achieves a Success Rate (SR) of 85.0% and a Progress Success Rate (PSR) of 94.4%, exceeding the strongest open-source baseline by 15% in SR. These results demonstrate that unified World Action Modeling together with progressive embodied pretraining effectively transforms large-scale embodied experience into generalizable robot manipulation capabilities.",
    "submittedDate": "2026-08-27",
    "primaryCategory": "Evaluation / Survey / Theory",
    "secondaryCategories": [],
    "bibtexKey": "Riemann10",
    "arxivUrl": "https://arxiv.org/abs/2608.27033",
    "codeUrls": [],
    "projectUrl": "https://riemann-dynamics.github.io/Riemann-1.0-Website",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.27033",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "泛化与动作对齐",
      "记忆与长时序"
    ],
    "architecture": "One Model",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.26103",
    "title": "Zero-WAM: In-Context World-Action Modeling from Human Videos for Open-Ended Task Generalization",
    "authors": "Zhou, Jiaming; Zhang, Qihang; Xu, Gangwei; Fan, Cunxin; Zhao, Yujie; Wang, Ruilin; Luo, Yiming; Yang, Shuai; Zhu, Xing; Shen, Yujun; Liang, Junwei; Xu, Yinghao",
    "affiliations": "Not identified",
    "contribution": "To achieve cross-task generalization, we bring this paradigm to robotic manipulation, and argue that the natural task specification for manipulation is a human video: unlike language, it provides rich visual cues about the intended task evolution. We present Zero-WAM, a causal video-action model that executes unseen tasks by following in-context human video guidance.",
    "abstract": "Zero-shot cross-task generalization, where a policy must execute manipulation tasks never seen during training, remains a central challenge in robot learning. In large language models, a novel task can be performed simply by specifying it in the context, without any parameter update. This form of in-context learning (ICL) turns generalization into a problem of task specification. To achieve cross-task generalization, we bring this paradigm to robotic manipulation, and argue that the natural task specification for manipulation is a human video: unlike language, it provides rich visual cues about the intended task evolution. We present Zero-WAM, a causal video-action model that executes unseen tasks by following in-context human video guidance. To address the scarcity of task-rich paired human-robot data, we propose an automatic pipeline that converts task-sampled robot trajectories into semantically matched human videos, yielding HumanGen, a dataset of 74.2K human-robot ICL pairs across 8.6K tasks. For model training, we further introduce an in-context future chunk prediction (IFP) objective that suppresses shortcuts learned from seen tasks and forces the policy to draw task information from the video prompt. On seven unseen tasks in RoboTwin 2.0 simulation, Zero-WAM achieves a 47.0% average success rate, an absolute improvement of 29.5 percentage points over the strongest video-action baseline. In real-world evaluations, it follows human video guidance to generalize to unseen task configurations involving multi-object scenes, long-horizon manipulation, and fine-grained insertion.",
    "submittedDate": "2026-08-27",
    "primaryCategory": "General WAM",
    "secondaryCategories": [],
    "bibtexKey": "ZeroWAM",
    "arxivUrl": "https://arxiv.org/abs/2608.26103",
    "codeUrls": [],
    "projectUrl": "https://robbyant-research.github.io/Zero-WAM/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.26103",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "视觉规划与IDM",
      "泛化与动作对齐",
      "记忆与长时序"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2606.26663",
    "title": "Tactile-WAM: Touch-Aware World Action Model with Tactile Asymmetric Attention",
    "authors": "Wu, Siyu; You, Linjing; Zhu, Junjie; Liu, Yaozu; Kaixiang, Huang; Yonghang, Chen; Li, Jituo; Zhang, Changhao; Liu, Jian; Chu, Hengshuo; Li, Qi; Zhao, Hengshuang",
    "affiliations": "2Institute of Automation, Chinese Academy of Sciences; 3The University of Hong Kong; 4Zhejiang University",
    "contribution": "On five real-robot tasks, Tactile-WAM achieves 49.2% success.",
    "abstract": "World Action Models (WAMs) jointly predict future visual observations and actions, but visual futures alone often miss slip, jamming, contact-direction changes, and subtle misalign- ment in contact-rich manipulation. Tactile signals reveal these hidden physical states, yet naive tactile-token injection can disrupt visual dynamics modeling due to the limited scale of tactile data, a phenomenon we term tactile pollution. We in- troduce Tactile-WAM, which uses asymmetric attention to block video queries from tactile keys while preserving tac- tile access for action queries. A contact-change-aware bias further strengthens action attention to touch. Because tactile pixel changes do not reliably reflect contact changes, we derive Observed proxy changes drive the attention bias, while future- proxy supervision preserves action-relevant contact dynamics in predicted tactile representations. On ManiFeel, visual-path isolation reduces deviation from the RGB-only trajectory by 21.8% in MSE at the step-matched 20K checkpoint without a statistically detectable change in ground-truth video qual- ity. The full model improves average success from 15.6% to 32.7%, with VideoClean providing the largest gain. On five real-robot tasks, Tactile-WAM achieves 49.2% success.",
    "submittedDate": "2026-08-27",
    "primaryCategory": "Multimodal / Tactile WAM",
    "secondaryCategories": [],
    "bibtexKey": "TactileWAM",
    "arxivUrl": "https://arxiv.org/abs/2606.26663",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2606.26663",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "联合视频动作建模",
      "多模态触觉音频"
    ],
    "architecture": "One Model",
    "predictionParadigm": "联合预测",
    "quadrant": "Q1 · One Model × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.25956",
    "title": "4DGS-WAM: Bridging Past and Future with an Object-Centric World Action Model based on 4D Gaussian Splatting",
    "authors": "Ma, Yueen; Xu, Zenglin; King, Irwin",
    "affiliations": "The Chinese University of Hong Kong; Shanghai Academy of AI for Science; Fudan University",
    "contribution": "These models can achieve exceptional visual quality, but they lack explicit spatial structure for individual objects and repeatedly process redundant background content.",
    "abstract": "Current world action models (WAMs) typically operate on 2D visual data. These models can achieve exceptional visual quality, but they lack explicit spatial structure for individual objects and repeatedly process redundant background content. Although point clouds can represent the world in 3D space, they can be difficult to align and accumulate across viewpoints. In this paper, we leverage an explicit 4D Gaussian Splatting (4DGS) representation that separately models dynamic objects and the static background of a scene. For dynamic objects, we use a policy model to predict future actor actions and a world model to predict transformations of their observed Gaussian splats. The static background need not be regenerated for future states, as much of it has already been observed in past frames. This forms an object-centric world action model, which we name 4DGS-WAM. It lifts 2D observations into a persistent 4D representation so that previously observed static content can be reused during future prediction. Future-state extrapolation can then focus on modeling the evolution of dynamic objects. Experiments on KITTI-MOT evaluate short-horizon prediction and past reconstruction.",
    "submittedDate": "2026-08-26",
    "primaryCategory": "3D/4D WAM",
    "secondaryCategories": [],
    "bibtexKey": "4DGSWAM",
    "arxivUrl": "https://arxiv.org/abs/2608.25956",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.25956",
    "pdfUrl": "https://arxiv.org/pdf/2608.25956",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{4DGSWAM,\n  title={4DGS-WAM: Bridging Past and Future with an Object-Centric World Action Model based on 4D Gaussian Splatting},\n  author={Ma, Yueen and Xu, Zenglin and King, Irwin},\n  journal={arXiv:2608.25956},\n  year={2026}\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "自动驾驶",
      "三维表示与状态估计"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.07468",
    "title": "SimWAM: A Simple World Action Model for End-to-End Autonomous Driving",
    "authors": "Zhao, Zongchuang; Zhou, Xin; Xu, Tianyang; Sun, Zhengyang; Zhou, Kaixuan; Wu, Yu; Li, Honglin; Liang, Dingkang; Bai, Xiang",
    "affiliations": "1 Huazhong University of Science & Technology, 2 Dongfeng Research & Development Institute",
    "contribution": "We present SimWAM, a simple yet effective WAM that leverages future-video prediction as a training-time supervision signal. Our SimWAM achieves 91.5 PDMS on NAVSIM, surpasses state-of-the-art WAM-based planners with substantially lower latency, and transfers zero-shot to nuScenes.",
    "abstract": "World-Action Models (WAMs) improve end-to-end autonomous driving by transferring video dynamics priors to action prediction, but existing methods incur costly test-time future imagination. We present SimWAM, a simple yet effective WAM that leverages future-video prediction as a training-time supervision signal. It co-trains a pretrained video expert and a lightweight action expert with joint flow matching. An isolated attention mask keeps action prediction independent of future frames, allowing trajectory prediction without explicit future-frame generation at inference. Since the two experts share no parameters and interact only through a unified attention interface, the video backbone could be replaced and the action expert scaled independently without modifying the learning objective or inference pipeline. We further apply reinforcement learning to optimize a compositional driving reward beyond trajectory imitation. Our SimWAM achieves 91.5 PDMS on NAVSIM, surpasses state-of-the-art WAM-based planners with substantially lower latency, and transfers zero-shot to nuScenes. These results position SimWAM as a simple yet solid baseline that could readily benefit from advances in video generation for efficient autonomous driving. The code and model weights are available at https://github.com/H-EmbodVis/SimWAM/.",
    "submittedDate": "2026-08-26",
    "primaryCategory": "Navigation / Driving / Domain WAM",
    "secondaryCategories": [
      "Real-Time / Efficient WAM",
      "WAM + RL"
    ],
    "bibtexKey": "SimWAM",
    "arxivUrl": "https://arxiv.org/abs/2608.07468",
    "codeUrls": [
      "https://github.com/H-EmbodVis/SimWAM"
    ],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.07468",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "自动驾驶",
      "高效推理与实时控制",
      "策略后训练与WM-RL"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2607.00836",
    "title": "From World Models to World Action Models: A Concise Tutorial for Robotics",
    "authors": "Zhang, Xiaoxiong; Zeng, Xiong; Zhang, Wei",
    "affiliations": "School of Automation and Intelligent Manufacturing, Southern University of Science and Technology, Shenzhen, China; LimX Dynamics, Shenzhen, China",
    "contribution": "Rather than providing an exhaustive survey, this paper presents a concise tutorial on world models and world action models for robotics.",
    "abstract": "Rather than providing an exhaustive survey, this paper presents a concise tutorial on world models and world action models for robotics. After reading the tutorial, readers should have a clear understanding of what constitutes a \"world\", how world models and world action models are defined, and what roles they play within robotic AI systems. The tutorial also develops a unified perspective for comparing representative approaches, such as World Labs' spatial intelligence models, Yann LeCun's JEPA framework, and NVIDIA's Cosmos platform, and clarifies how these models differ in their representations, predictive capabilities, and interaction mechanisms.",
    "submittedDate": "2026-08-26",
    "primaryCategory": "Latent / Representation WAM",
    "secondaryCategories": [
      "Evaluation / Survey / Theory"
    ],
    "bibtexKey": "CTR",
    "arxivUrl": "https://arxiv.org/abs/2607.00836",
    "codeUrls": [
      "https://github.com/clearlab-sustech/WorldModelSurvey"
    ],
    "projectUrl": "https://clearlab-sustech.github.io/WorldModelSurvey/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2607.00836",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "Related resources",
    "subcategories": [
      "综述与技术资源"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2605.27947",
    "title": "SANTS: A State-Adaptive Scheduler for World Action Models",
    "authors": "Sun, Yirui; Zhuge, Guangyu; Liu, Keliang; Gu, Jie; Dai, Shiqin; Bing, Xinyu; Gan, Zhongxue; Tian, Chunxu",
    "affiliations": "1Fudan University; Project page:",
    "contribution": "We introduce State-Adaptive Noise Trajectory Scheduler (SANTS), a lightweight scheduler for video-to-action diffusion policies.",
    "abstract": "World Action Models (WAMs) improve robot manipulation by using video-based future representations to condition action generation. In pixel-space WAMs, however, the best action condition is not necessarily the fully denoised video. Controlled denoising-depth scans show that video refinement can reduce action error up to a state-dependent point, after which the gain may saturate or even reverse when late predictions become less action-relevant or physically unreliable. This suggests that action generation should use a state-dependent point along the video noise trajectory rather than a fixed terminal denoising depth. We introduce State-Adaptive Noise Trajectory Scheduler (SANTS), a lightweight scheduler for video-to-action diffusion policies. At each video decision point, SANTS reads the current video-state representation and noise level, then jointly predicts a cumulative stopping hazard and a relative noise-progression ratio. SANTS is post-trained with a path-level reward computed after the frozen action branch generates the final action chunk, so the scheduler is optimized for downstream action quality rather than intermediate video fidelity, while redundant video-state updates are explicitly penalized. Experiments show that SANTS reaches 94.4%94.4\\% overall success on RoboTwin 2.0 and 73.1%73.1\\% average success across seven real-robot tasks, while reducing latency by 81.7%81.7\\% and 79.0%79.0\\% relative to full video denoising, respectively. These results indicate that adaptive selection along the video noise trajectory can preserve the control benefits of WAM-style future reasoning while removing much of its redundant inference cost.",
    "submittedDate": "2026-08-26",
    "primaryCategory": "3D/4D WAM",
    "secondaryCategories": [
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "SANTS",
    "arxivUrl": "https://arxiv.org/abs/2605.27947",
    "codeUrls": [],
    "projectUrl": "https://advanced-robotics-lab.github.io/SANTS/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2605.27947",
    "pdfUrl": "https://arxiv.org/pdf/2605.27947",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{SANTS,\n  title={SANTS: A State-Adaptive Scheduler for World Action Models},\n  author={Sun, Yirui and Zhuge, Guangyu and Liu, Keliang and Gu, Jie and Dai, Shiqin and Bing, Xinyu and Gan, Zhongxue and Tian, Chunxu},\n  journal={arXiv:2605.27947},\n  year={2026}\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "视觉规划与IDM",
      "高效推理与实时控制"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.26200",
    "title": "GameWAM: A World Action Model for Video Games",
    "authors": "Guo, Yuncheng; Zhang, Zhanqiu; Guo, Yiwen; Li, Weijia",
    "affiliations": "1 Fudan University; 4 Tsinghua Shenzhen International Graduate School",
    "contribution": "We introduce GameWAM, to our knowledge the first WAM for native closed-loop gameplay and GUI control.",
    "abstract": "Modern video games combine first-person perception, rapid visual changes, persistent world state, and heterogeneous native controls. Existing game agents map visual and task context directly to actions but lack explicit world dynamics modeling, whereas interactive game world models predict visual futures from supplied actions but do not serve as task policies. World-Action Models (WAMs) unify these objectives, but remain largely unexplored under the dynamics and open-ended interaction of video games. We introduce GameWAM, to our knowledge the first WAM for native closed-loop gameplay and GUI control. GameWAM jointly generates future visual observations and executable keyboard-mouse trajectories through parallel visual and action generative processes with block-causal conditioning and flow matching. To support joint world-action learning, we construct synchronized gameplay and GUI trajectories. To handle heterogeneous native control, GameWAM predicts a gameplay/GUI mode at each action step and generates actions with mode-specific prediction distributions and continuous-action normalization. For long-horizon interaction, block-cycle control predicts beyond the committed horizon, executes only a short action prefix, and replans from new observations, while fine-grained within-cycle context and hierarchical cross-cycle history preserve temporal continuity. Experiments demonstrate competitive task success with fewer executed native actions than the compared agents. We further uncover Low-Frequency Action Source Imprinting (LASI), in which low-frequency components of the sampled action source systematically steer coarse generated camera motion under fixed conditioning, revealing a source-sensitivity failure mode in generative control. Project page is available at https://yunncheng.github.io/GameWAM/.",
    "submittedDate": "2026-08-25",
    "primaryCategory": "Navigation / Driving / Domain WAM",
    "secondaryCategories": [],
    "bibtexKey": "GameWAM",
    "arxivUrl": "https://arxiv.org/abs/2608.26200",
    "codeUrls": [],
    "projectUrl": "https://yunncheng.github.io/GameWAM/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.26200",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "联合视频动作建模",
      "记忆与长时序"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "联合预测",
    "quadrant": "Q3 · Dual-system × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.24714",
    "title": "GaussianWAM: Distilling Geometry and Semantics from 3D Gaussian Fields into World-Action Models",
    "authors": "Zhang, Zijian; Jiang, Yuqing; Zhou, Weitao; Li, Minglei; Zhang, Jinhao; Mu, Yao; Li, Xiaofan; Zhao, Hao; Yu, Haibao",
    "affiliations": "1Tuojing Intelligence, 2University of Chinese Academy of Sciences,3Institute of Automation, Chinese; Academy of Sciences,4Tsinghua University,5Simple AI, 6Harbin Institute of Technology (Shenzhen); 7Shanghai Jiao Tong University,8Zhejiang University, 9Institute for AI Industry Research (AIR), Tsinghua; University, 10The University of Hong Kong",
    "contribution": "We propose \\textbf{GaussianWAM}, a training-time representation-enhancement framework that organizes geometric and semantic supervision through a 3D Gaussian field.",
    "abstract": "World-Action Models (WAMs) jointly learn future visual prediction and action generation, using video dynamics as a representation-learning signal for robotic manipulation. However, their video latents are primarily optimized for visual prediction and are not explicitly encouraged to preserve cross-view geometric structure or spatially localized, object-relevant semantics. We propose \\textbf{GaussianWAM}, a training-time representation-enhancement framework that organizes geometric and semantic supervision through a 3D Gaussian field. Given synchronized multi-view observations, frozen geometry and vision foundation models provide depth, camera parameters, and dense semantic features. GaussianWAM binds these heterogeneous signals to shared Gaussian primitives and renders spatially aligned semantic, depth, and coverage targets, which are distilled into the current-observation representations of the WAM. All teacher models, Gaussian components, and auxiliary prediction heads are removed after training, leaving the original WAM inference path without additional modules or forward computation. On LIBERO-Plus, GaussianWAM improves FastWAM from 52.05\\% to 71.29\\% and Cosmos Policy from 71.52\\% to 77.30\\%. Direct CLIP and VGGT distillation already establishes a strong FastWAM baseline of 69.37\\%, while Gaussian-field unification further improves it to 71.29\\%, supporting the benefit of spatially organizing heterogeneous teacher signals. GaussianWAM also improves performance on standard LIBERO and shows positive transfer trends on RoboTwin and real-world manipulation. These results suggest that training-time Gaussian distillation provides a practical way to inject geometry- and semantics-related supervision into WAM representations without changing their deployment architecture.",
    "submittedDate": "2026-08-25",
    "primaryCategory": "3D/4D WAM",
    "secondaryCategories": [
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "GaussianWAM",
    "arxivUrl": "https://arxiv.org/abs/2608.24714",
    "codeUrls": [],
    "projectUrl": "https://tuojingai.github.io/GaussianWAM-project-page/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.24714",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "三维多视角建模",
      "泛化与动作对齐"
    ],
    "architecture": "不适用",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.23486",
    "title": "GeoWAM: Visual Geometry World Action Models for Autonomous Driving",
    "authors": "Lu, Yiren; Ye, Xin; Liu, Jiaming; Jacobson, Philip; Yao, Jin; Chen, Yi-chung; Merino, Liam; Kurra, Dhruva Dixith; Cai, Min; Lampo, Tom; Yin, Yu; Guo, Danhua; Yaman, Burhan",
    "affiliations": "1Uber AV Labs 2Case Western Reserve University",
    "contribution": "Building on this insight, we introduce \\textbf{GeoWAM}, a visual geometry world action model for autonomous driving.",
    "abstract": "World action models (WAMs) have recently gained increasing attention as a framework for jointly modeling scene evolution and ego actions in autonomous driving. Most existing WAMs learn scene dynamics in pixel space by combining a video-generation backbone for future-observation prediction with an action head for ego-trajectory prediction. Pixels, however, provide only an indirect representation of these dynamics: they entangle geometry and motion with appearance, texture, and illumination, forcing the model to infer three-dimensional transformations from two-dimensional observations. We argue that geometry, represented by point clouds, offers a more natural state space for driving because it explicitly captures spatial structure and the rigid and non-rigid transformations that govern scene evolution while directly aligning with the space in which driving actions are executed. Building on this insight, we introduce \\textbf{GeoWAM}, a visual geometry world action model for autonomous driving. Rather than predicting future images, GeoWAM is pretrained to forecast future scene geometry, yielding representations that jointly encode spatial structure and temporal evolution. A geometry-conditioned action head then leverages these learned geometric dynamics to predict future ego trajectories. Extensive open-loop and closed-loop evaluations show that visual geometry world modeling yields substantially stronger driving policies than image-based alternatives, establishing future-geometry prediction as an effective pretraining objective for autonomous driving.",
    "submittedDate": "2026-08-25",
    "primaryCategory": "3D/4D WAM",
    "secondaryCategories": [
      "Navigation / Driving / Domain WAM"
    ],
    "bibtexKey": "GeoWAM",
    "arxivUrl": "https://arxiv.org/abs/2608.23486",
    "codeUrls": [],
    "projectUrl": "https://yiren-lu.com/project_pages/geowam/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.23486",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "三维多视角建模",
      "自动驾驶",
      "视觉规划与IDM"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.23927",
    "title": "GlanceWAM: Sparse Test-Time Imagination for World-Action Models",
    "authors": "Wang, Linhan; An, Zijian; Zhang, Mingyuan; Dai, Chen; Xu, Yi; Cui, Can; Yang, Zichong; Chen, Yinlin; Zhou, Lifeng; Lu, Chang-Tien",
    "affiliations": "1 Virginia Tech 2 Drexel University 3 Northeastern University 4 Purdue University",
    "contribution": "We show that visual imagination achieves both real-time inference and superior success rates when generated asynchronously off the critical path and consumed directly in latent space. We introduce GlanceWAM, which decouples imagination from control within a single video DiT: an asynchronous proposer glances ahead on a slow clock to imagine a single lookahead frame seconds into the future in the background, while an action head decodes action chunks at control rate (48 ms) purely in latent space without blocking.",
    "abstract": "Video generative models provide rich physical priors for robot learning, yet existing world-action models (WAMs) face a fundamental trade-off: synchronous video generation at control rate is latency-prohibitive, while abandoning test-time visual imagination sacrifices task success. We show that visual imagination achieves both real-time inference and superior success rates when generated asynchronously off the critical path and consumed directly in latent space. We introduce GlanceWAM, which decouples imagination from control within a single video DiT: an asynchronous proposer glances ahead on a slow clock to imagine a single lookahead frame seconds into the future in the background, while an action head decodes action chunks at control rate (48 ms) purely in latent space without blocking. Enabled by a non-interfering attention mask that isolates video representations and staleness-robust horizon training that accommodates asynchronous lookahead aging, GlanceWAM breaks the speed-success dilemma. Trained purely on demonstrations, it attains 72.2% on the 24-task RoboCasa kitchen benchmark (surpassing synchronous Cosmos Policy at 67.1% and imagination-free co-training at 64.4%) and 99.0% on LIBERO, executing at 48 ms per chunk on an NVIDIA A100 GPU (24x faster than synchronous baselines). Code is available at https://github.com/linhanwang/GlanceWAM.",
    "submittedDate": "2026-08-24",
    "primaryCategory": "Latent / Representation WAM",
    "secondaryCategories": [
      "Evaluation / Survey / Theory",
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "GlanceWAM",
    "arxivUrl": "https://arxiv.org/abs/2608.23927",
    "codeUrls": [
      "https://github.com/linhanwang/GlanceWAM"
    ],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.23927",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "视觉规划与IDM",
      "高效推理与实时控制",
      "记忆与长时序"
    ],
    "architecture": "One Model",
    "predictionParadigm": "IDM",
    "quadrant": "Q2 · One Model × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2606.27504",
    "title": "ReWorld: Representation Learning for World Action Models",
    "authors": "Xia, Tianze; Zhou, Lijun; Xiong, Kaixin; Yao, Jingfeng; Zhu, Zhenxin; Sun, Haiyang; Wang, Bing; Chen, Guang; Liu, Wenyu; Ye, Hangjun; Wang, Xinggang",
    "affiliations": "Tianze Xia, Jingfeng Yao, Wenyu Liu, and Xinggang Wang are with Huazhong University of Science and Technology, Wuhan, China. Tianze Xia, Lijun Zhou, Kaixin Xiong, Zhenxin Zhu, Haiyang Sun, Bing Wang, Guang Chen, and Hangjun Ye are with Xiaomi EV, China. Tianze Xia and Lijun Zhou contributed equally to this work. Lijun Zhou is the project lead. Corresponding author: Xinggang Wang (E-mail: ); IEEE Publication Technology Department",
    "contribution": "We present ReWorld, the first representation learning framework specifically designed for autonomous-driving WAMs.",
    "abstract": "World Action Models (WAMs) unify future environment prediction with action generation for autonomous driving, yet existing approaches optimize only the final outputs, leaving intermediate representations as incidental byproducts. We present ReWorld, the first representation learning framework specifically designed for autonomous-driving WAMs. ReWorld explicitly optimizes the latent world-to-action pathway through three complementary mechanisms. First, it imposes future-predictive supervision on intermediate Video DiT states to encode temporal scene dynamics, enabling self-guided sampling and a roughly twofold convergence speedup. Second, it aligns Action DiT states with their attended video readouts so that the retrieved world information is retained in the representations used for planning. Third, it shapes the action space using geometrically close yet low-scoring hard negatives to separate the expert trajectory from nearby unsafe alternatives. ReWorld constructs supervision entirely from the WAM's own generation targets and attended features, requiring no external encoders or teacher models and introducing only 0.3% additional per-step training cost. Experiments show that ReWorld reduces FVD from 81.3 to 61.9 on nuScenes, improves closed-loop PDMS from 89.1 to 90.4 on NAVSIM without reinforcement learning or test-time scoring, and increases frozen linear-probe accuracy from 68.3% to 80.2% on UCF-101 action recognition. These results indicate that explicitly optimized representations are central to translating world knowledge into planning capability in WAMs.",
    "submittedDate": "2026-08-24",
    "primaryCategory": "Latent / Representation WAM",
    "secondaryCategories": [
      "Navigation / Driving / Domain WAM",
      "WAM + RL"
    ],
    "bibtexKey": "ReWorld",
    "arxivUrl": "https://arxiv.org/abs/2606.27504",
    "codeUrls": [
      "https://github.com/xiaomi-research/ReWorld"
    ],
    "projectUrl": "https://xiaomi-research.github.io/reworld/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2606.27504",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "自动驾驶",
      "视觉规划与IDM",
      "泛化与动作对齐"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.22403",
    "title": "LD4WAM: Learning Latent Dynamics from Human Videos for World Action Models",
    "authors": "Shen, Zhenhao; Liang, Jiaqi; Lu, Jasper; Jiang, Feng; Wang, Yuran; Wei, Chuanbo; Liu, Jiayi; Yang, Jianchun; Yu, Qize; You, Jiadi; Hao, Ce; He, Guanqi; Xie, Chen; Wu, Ruihai",
    "affiliations": "1Peking University 2WUJI 3Lightwheel 4Beijing Zhongguancun Academy; 5Wuhan University 6The University of Hong Kong",
    "contribution": "Human video is playing an increasingly central role in training World Action Models (WAMs), owing to its diversity and low collection cost relative to teleoperated robot data. However, most WAMs learn from such video only by predicting pixel-level future frames, giving dynamics that are not directly actionable, whereas motion retargeting recovers directly actionable actions but leaves a large visual gap across embodiments.",
    "abstract": "Human video is playing an increasingly central role in training World Action Models (WAMs), owing to its diversity and low collection cost relative to teleoperated robot data. However, most WAMs learn from such video only by predicting pixel-level future frames, giving dynamics that are not directly actionable, whereas motion retargeting recovers directly actionable actions but leaves a large visual gap across embodiments. We therefore propose motion-aligned latent dynamics as an embodiment-agnostic representation to bridge video priors and low-level actions. We further present LD4WAM, which pairs a Latent Dynamics Model trained with semantic reconstruction and real motion alignment with a World Dynamics Action Model built as a mixture-of-transformers (MoT), which preserves full future-video generation and uses learnable queries to distill these latent dynamics from generated futures for action conditioning. Pretrained on our curated unified dataset of over 5{,}000 hours of human and robot data, LD4WAM performs strongly in RoboTwin simulation and on real robots equipped with both grippers and dexterous hands, while generalizing well to unseen objects and backgrounds.",
    "submittedDate": "2026-08-23",
    "primaryCategory": "Latent / Representation WAM",
    "secondaryCategories": [],
    "bibtexKey": "LD4WAM",
    "arxivUrl": "https://arxiv.org/abs/2608.22403",
    "codeUrls": [
      "https://github.com/stubborn111/LD4WAM"
    ],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.22403",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "潜空间预测与JEPA",
      "视觉规划与IDM",
      "泛化与动作对齐"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.22364",
    "title": "WAM-OPD: On-Policy Distillation for World Action Models",
    "authors": "Yang, Liuhaichen; Jiang, Zhuang; Sheng, Chenchao; Tang, Zezhi",
    "affiliations": "1Department of Computer Science, University College London; 2Department of Mechanical Engineering, University College London",
    "contribution": "We introduce WAM-OPD, a deployment-consistent post-training recipe for a video-first WAM.",
    "abstract": "World action models (WAMs) couple visual future prediction with robot action generation, but accelerated students can lose task capabilities during distillation and later encounter states that are poorly represented by offline data. We study whether on-policy distillation (OPD) can repair such a student without requiring sparse-reward reinforcement learning. We introduce WAM-OPD, a deployment-consistent post-training recipe for a video-first WAM. The student acts in the environment and therefore determines the history distribution. A frozen teacher labels those student histories with coherent video and action targets, while the student action branch is trained under its own generated video plan, as it is at deployment. Joint video and action losses update lightweight adapters in the shared backbone, together with an action flow-matching regularizer. In preliminary RoboTwin 2.0 studies on two tasks, the released one-video/one-action-step Flash-WAM improves from 0.0% to 58.3% success on HANDOVER MIC, and from 16.7% to 33.3% on PUT OBJECT CABINET. These task-specific results are an initial capability proof rather than evidence of broad or uniform generalization. They nevertheless suggest that dense teacher supervision on student-induced histories is a promising post-training interface for video-first WAMs.",
    "submittedDate": "2026-08-23",
    "primaryCategory": "WAM + RL",
    "secondaryCategories": [
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "WAMOPD",
    "arxivUrl": "https://arxiv.org/abs/2608.22364",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.22364",
    "pdfUrl": "https://arxiv.org/pdf/2608.22364",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{WAMOPD,\n  title={WAM-OPD: On-Policy Distillation for World Action Models},\n  author={Yang, Liuhaichen and Jiang, Zhuang and Sheng, Chenchao and Tang, Zezhi},\n  journal={arXiv:2608.22364},\n  year={2026}\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "策略后训练与WM-RL",
      "视觉规划与IDM"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.20735",
    "title": "ForeTime-VLA: Causal Future-Token Distillation from a World Action Model for Conveyor-Belt Manipulation",
    "authors": "Ma, Siyuan; Zhang, Yutian; Zhang, Boshi; Wu, Qinglian; Zhai, Jiaqi; Wei, Dong; Huang, Xiaojin",
    "affiliations": "Tsinghua University, Beijing, China; Shanghai Artificial Intelligence Laboratory, Shanghai, China; Harbin Institute of Technology, Harbin, China; Hangzhou Yunshenchu Technology Co., Ltd. (DEEP Robotics), Hangzhou, China",
    "contribution": "We introduce ForeTime-VLA, a dense pi0.5 policy that distills a future-aware, action-equivalent representation from a frozen Fast-WAM-derived teacher while remaining causal at inference. In quantitative real-robot evaluation, ForeTime-VLA achieves 81.1% stationary and 58.9% slow-moving grasp success, exceeding the next-best reference by 12.2 and 22.2 percentage points, respectively.",
    "abstract": "Manipulating moving objects requires a policy to anticipate contact events, yet vision-language-action (VLA) policies are commonly fine-tuned from the current observation alone. World action models (WAMs) learn predictive dynamics, but running a video-scale teacher or explicitly imagining future frames at deployment is costly. We introduce ForeTime-VLA, a dense pi0.5 policy that distills a future-aware, action-equivalent representation from a frozen Fast-WAM-derived teacher while remaining causal at inference. Offline, current and future video latents are compressed into a whitened 64-D target. Online, an eight-frame history encoder predicts this target together with manipulation phase and normalized time-to-transition. Four future tokens and one phase token condition the VLM prefix, while the predicted future and transition horizon condition the action expert. Training retains the original flow-matching action target and adds cosine, relational geometry, phase, time-to-transition, and action-equivalence objectives. On a deduplicated conveyor-belt dataset, we compare 40k-step checkpoints on 768 matched windows per split. Test MAE decreases from 0.134119 to 0.130593 (2.63%; paired-bootstrap 95% CI: 0.82-4.48% improvement), and test L2 decreases by 3.02%, at a 2.46-2.93% latency cost. In quantitative real-robot evaluation, ForeTime-VLA achieves 81.1% stationary and 58.9% slow-moving grasp success, exceeding the next-best reference by 12.2 and 22.2 percentage points, respectively. Across three belt speeds, it completes 44/90 grasps versus 23/90 for pi0.5, including 11/30 versus 2/30 at fast speed. The agreement between offline orientation gains and reduced real-robot contact-pose failures supports causal future-token distillation as an effective way to improve dynamic manipulation without deploying the world-model teacher.",
    "submittedDate": "2026-08-23",
    "primaryCategory": "Real-Time / Efficient WAM",
    "secondaryCategories": [
      "3D/4D WAM",
      "Evaluation / Survey / Theory"
    ],
    "bibtexKey": "ForeTimeVLA",
    "arxivUrl": "https://arxiv.org/abs/2608.20735",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.20735",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "潜空间预测与JEPA",
      "视觉规划与IDM",
      "泛化与动作对齐"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2607.28391",
    "title": "TacWAM: Anchor-Guided World Action Model with Mechanics-Aware Tactile Prediction",
    "authors": "Jin, Lei; Ma, Yiding; Zhang, Xin; Gao, Chen; Wu, Wei; Li, Yong",
    "affiliations": "1Tsinghua University 2Manifold AI",
    "contribution": "We present TacWAM, a mechanics-aware tactile WAM that addresses this challenge in three steps. TacWAM achieves an average success rate of 75.0%, exceeding the strongest evaluated baseline by 37.5 percentage points.",
    "abstract": "World Action Models (WAMs) combine future-state prediction with robot action generation, but existing approaches largely rely on visual futures. Visual prediction captures scene structure and object motion, yet provides limited supervision for force, deformation, shear, and slip during contact-rich manipulation. This creates two design requirements: tactile futures should carry meaningful physical information, and they should not become privileged cues for action generation. We present TacWAM, a mechanics-aware tactile WAM that addresses this challenge in three steps. First, a Spatially Aligned Fusion (SAF) Tactile Encoder maps tactile appearance, dense force fields, and deformation flow into a shared latent prediction space, with bilateral force and torque reconstruction preserving global contact information. Second, a tactile history encoder provides temporal context so future tactile prediction reflects how force and deformation change beyond the current tactile observation. Third, Anchor-Guided Tri-Modal (AGT) Attention separates current visual and tactile anchors, future prediction tokens, and action tokens, allowing future tactile states to supervise training without being directly read by the action branch. We evaluate TacWAM on four real-world contact-rich manipulation tasks covering fragile grasping, sustained surface contact, and dynamic in-hand manipulation. TacWAM achieves an average success rate of 75.0%, exceeding the strongest evaluated baseline by 37.5 percentage points. Staged ablations show consistent degradation when tactile history is removed and access to future prediction targets is relaxed. These results indicate that future tactile supervision can improve contact-aware action learning when combined with informative tactile representations and deployment-consistent information constraints.",
    "submittedDate": "2026-08-23",
    "primaryCategory": "Multimodal / Tactile WAM",
    "secondaryCategories": [
      "Latent / Representation WAM"
    ],
    "bibtexKey": "TacWAM",
    "arxivUrl": "https://arxiv.org/abs/2607.28391",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2607.28391",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "多模态触觉音频",
      "高效推理与实时控制"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.22197",
    "title": "On the Capability Separation Between World-Model Policy Learning and Imitated World-Action Models",
    "authors": "Yu, Yang",
    "affiliations": "Nanjing University",
    "contribution": "World-action models predict a future outcome and then infer an associated action. Although this factorization can improve representation learning and data efficiency, it is unclear whether it provides stronger control capability than direct behavior cloning when both are trained from the same observational demonstrations.",
    "abstract": "World-action models predict a future outcome and then infer an associated action. Although this factorization can improve representation learning and data efficiency, it is unclear whether it provides stronger control capability than direct behavior cloning when both are trained from the same observational demonstrations. We compare a direct behavior-cloning policy, an imitation-trained world-action policy, and a policy optimized with an action-conditioned world model. At the controller-class level, every world-action policy can be flattened into a direct stochastic policy with the same closed-loop trajectory distribution. At the population level, under realizability, exact optimization, common deployment information, and distribution-preserving deployment, direct behavior cloning and world-action imitation both recover the observational behavior policy. Thus, future prediction changes the learning factorization but not the unrestricted external policy class or ideal imitation target. Action-conditioned world-model learning differs by predicting outcomes under specified actions and comparing them through a control objective. We characterize the irreducible action-specific prediction error of future models that do not condition on the candidate action, identify conditions under which a world-action joint can recover an interventional forward model, and show that observational demonstrations do not identify action effects in general. Finally, we construct an environment family in which every observational learner has positive worst-case regret, whereas one informative intervention permits zero regret. The key distinction is therefore between predicting futures associated with observed behavior and predicting consequences of specified actions for policy optimization.",
    "submittedDate": "2026-08-22",
    "primaryCategory": "WAM + RL",
    "secondaryCategories": [
      "Latent / Representation WAM",
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "CSBPLI",
    "arxivUrl": "https://arxiv.org/abs/2608.22197",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.22197",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "Related resources",
    "subcategories": [
      "理论与规划",
      "经典WM与模型式RL"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.20974",
    "title": "WA-JEPA: Rethinking the Video JEPA Paradigm for World-Action Modeling in Autonomous Driving",
    "authors": "Wang, Xinlin; Xiang, Yujiao; Zhou, Yuheng; Wang, Jingqi; Huang, Minqing; Huang, Jiajie; Wei, Dongxu; Zhou, Tingguang; Wang, Xiyang; Chen, Gong; Xu, Zhi; Tan, Feiyang; Zhou, Hangning; Yang, Mu",
    "affiliations": "1Afari Intelligent Drive 2University of Electronic Science and Technology of China; 3Southeast University 4Beijing University of Posts and Telecommunications5Tianjin University",
    "contribution": "Video Joint Embedding Predictive Architecture (V-JEPA) learns powerful spatiotemporal representations from video through self-supervised latent feature prediction. However, V-JEPA is built around random-mask completion and deterministic regression, making it fundamentally ill-suited for autonomous driving planning that demands future-directed prediction tightly coupled with action.",
    "abstract": "Video Joint Embedding Predictive Architecture (V-JEPA) learns powerful spatiotemporal representations from video through self-supervised latent feature prediction. However, V-JEPA is built around random-mask completion and deterministic regression, making it fundamentally ill-suited for autonomous driving planning that demands future-directed prediction tightly coupled with action. To address this, we rethink the V-JEPA paradigm and present WA-JEPA, a V-JEPA-native world-action model designed for autonomous driving planning. Instead of random spatiotemporal masking, WA-JEPA employs hybrid future-masked pre-training, where the model infers future latents from observed context. Departing from deterministic regression, we recast future prediction as conditional flow matching over latent futures, which substantially improves the model's ability to generate plausible future latents for downstream planning. Finally, a joint future-action predictor is proposed to denoise future scene tokens and ego trajectories together in a unified spatiotemporal latent space, allowing action supervision to directly shape planning-relevant world representations. Pre-trained on nuPlan videos and fine-tuned on NAVSIM, WA-JEPA reaches 91.7 EPDMS on NAVSIM-v2, surpassing the strongest end-to-end and world-action baselines by 1.6 and 1.3 EPDMS, and, without HUGSIM-specific fine-tuning, attains the best HD-Score of 0.4462 on the closed-loop HUGSIM benchmark under the same evaluation protocol. These results validate V-JEPA-native world-action modeling as a powerful and scalable paradigm for autonomous driving planning. Code is available at https://github.com/AFARI-Research/WA-JEPA.",
    "submittedDate": "2026-08-21",
    "primaryCategory": "Latent / Representation WAM",
    "secondaryCategories": [
      "Evaluation / Survey / Theory",
      "Navigation / Driving / Domain WAM"
    ],
    "bibtexKey": "WAJEPA",
    "arxivUrl": "https://arxiv.org/abs/2608.20974",
    "codeUrls": [
      "https://github.com/AFARI-Research/WA-JEPA"
    ],
    "projectUrl": "https://wa-jepa.github.io/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.20974",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "自动驾驶",
      "潜空间预测与JEPA",
      "联合视频动作建模"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "联合预测",
    "quadrant": "Q3 · Dual-system × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.20114",
    "title": "DECOWAM: Decoupled Whole-Body World-Action Model for Legged Mobile Manipulation",
    "authors": "Ma, Siyuan; Zhang, Boshi; Zhang, Yutian; Wu, Qinglian; Zhai, Jiaqi; Wei, Dong; Yu, Qiaojun",
    "affiliations": "Tsinghua University, Beijing, China Shanghai Artificial Intelligence Laboratory, Shanghai, China Harbin Institute of Technology, Harbin, China Hangzhou Yunshenchu Technology Co., Ltd. (DEEP Robotics), Hangzhou, China Equal contribution. Corresponding authors; Tsinghua University, Beijing, China; Shanghai Artificial Intelligence Laboratory, Shanghai, China; Harbin Institute of Technology, Harbin, China; Hangzhou Yunshenchu Technology Co., Ltd. (DEEP Robotics), Hangzhou, China",
    "contribution": "Here we introduce DECOWAM, a whole-body world-action model that separates these factors through dedicated conditional interfaces.",
    "abstract": "Mobile manipulation requires a robot to predict how locomotion and arm motion jointly alter future observations and control. Existing world-action models, developed largely for fixed-base platforms, do not explicitly distinguish camera ego-motion from base and arm actions. Here we introduce DECOWAM, a whole-body world-action model that separates these factors through dedicated conditional interfaces. DECOWAM freezes an adapted FastWAM backbone and trains residual adapters, an action-equivalent future bottleneck distilled from privileged observations, adversarially separated base and arm latents, and base-velocity conditioning for video prediction. We further introduce ARMDOG, a real-robot dataset that synchronizes video, whole-body state and action, and language. On a fixed replay protocol, DECOWAM improved both future-video and action prediction over FastWAM, reducing action MSE by 21.7% with 25.95M trainable adaptation parameters. Across 79 closed-loop trials per method, it achieved the highest observed whole-body coordination and base-displacement robustness among the compared systems, while task completion remained comparable to the strongest baseline. These results show that embodiment-aware factorization can support parameter-efficient joint visual prediction and whole-body control under moving viewpoints.",
    "submittedDate": "2026-08-21",
    "primaryCategory": "Navigation / Driving / Domain WAM",
    "secondaryCategories": [
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "DECOWAM",
    "arxivUrl": "https://arxiv.org/abs/2608.20114",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.20114",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "导航",
      "联合视频动作建模",
      "泛化与动作对齐"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.20284",
    "title": "Towards Surgical World-Action Modeling: A Preliminary Joint Visual-Trajectory Forecasting for Surgical Motion Planning",
    "authors": "Huang, Weiliang; Liu, Huanrong; Zhang, Bob; Dou, Qi; Chen, Zhen; Gu, Yun; Rosman, Guy; Li, Qingbiao",
    "affiliations": "1 Faculty of Information Science and Computing, University of Macau, Macau, China; 2 Faculty of Engineering, University of Macau, Macau, China; 3 University of Macau Advanced Research Institute in Hengqin, Zhuhai, China; 4 Department of Computer Science and Engineering, The Chinese University of Hong; 5 Department of Data Science and Artificial Intelligence, The Hong Kong Polytechnic; University, Hong Kong, China; 6 Shanghai Key Laboratory of Flexible Medical Robotics, Tongren Hospital, Institute; of Medical Robotics, Shanghai Jiao Tong University, Shanghai, China; 7 School of Automation and Intelligent Sensing, Shanghai Jiao Tong University; 8 School of Medicine, Duke University, Durham, North Carolina, USA",
    "contribution": "To bridge this gap, we present a preliminary joint visual-trajectory world-action model that simultaneously forecasts future visual states and instrument trajectories from historical surgical observations. The chunked strategy consistently outperforms direct one-shot prediction across all evaluated horizons, improving first-segment PSNR from 18.86 to 23.11 dB and reducing ADE from 45.77 to 22.22 pixels.",
    "abstract": "Reliable surgical planning requires models to anticipate not only how instruments will move, but also how the operative visual state will evolve together with such motion. Existing approaches typically treat future scene generation and instrument trajectory prediction as two separate tasks. Scene-only models cannot directly evaluate the accuracy of future instrument motion at the trajectory level, while trajectory-only models fail to capture the visual consequences of instrument movement, leaving the consistency between predicted trajectories and future scene evolution unaddressed. Jointly forecasting both provides a more complete account of surgical action-scene dynamics by enabling explicit trajectory-level evaluation while simultaneously modeling the corresponding visual evolution. To bridge this gap, we present a preliminary joint visual-trajectory world-action model that simultaneously forecasts future visual states and instrument trajectories from historical surgical observations. Specifically, we encode historical video frames and tool trajectories into latent representations, which are processed by a temporal-spatial encoder and subsequently decoded through separate visual-state and trajectory prediction heads. Based on this preliminary architecture, a chunked autoregressive rollout is repeatedly applied to predict fifteen future steps. The chunked strategy consistently outperforms direct one-shot prediction across all evaluated horizons, improving first-segment PSNR from 18.86 to 23.11 dB and reducing ADE from 45.77 to 22.22 pixels. These results demonstrate the initial feasibility of joint visual-motion forecasting. However, we observe progressive visual degradation and accumulated trajectory errors over longer prediction horizons, which remain important challenges for future surgical world-action modeling.",
    "submittedDate": "2026-08-20",
    "primaryCategory": "Navigation / Driving / Domain WAM",
    "secondaryCategories": [
      "Evaluation / Survey / Theory",
      "Latent / Representation WAM"
    ],
    "bibtexKey": "SMPJVTFS",
    "arxivUrl": "https://arxiv.org/abs/2608.20284",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.20284",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "联合视频动作建模"
    ],
    "architecture": "One Model",
    "predictionParadigm": "联合预测",
    "quadrant": "Q1 · One Model × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.19613",
    "title": "What Matters for Latent Actions in Robot Learning",
    "authors": "Xizhou Bu; Qingda Hu; Lei Zhou; Lingfeng Zhang; Yingbo Tang; Zihao Liu; Xinyi Tao; Zhiqiang Ma; Qingqiu Huang; Chufeng Tang; Hongbo Wang; Jing Zhang; Jiayi Ma; Hangjun Ye; Wei Li; Xiaoshuai Hao",
    "affiliations": "Fudan University; Tsinghua University; Shenzhen University of Advanced Technology; Sichuan University; Suzhou Evans Intelligent Technology Co., Ltd.; Morphi Intelligence Technology Co., Ltd.; Wuhan University; Xiaomi EV",
    "contribution": "This empirical study asks which video-derived latent actions improve robot policies. It compares modeling paradigms, regularization, and action integration under a shared training framework. Raw-frame LAPO remains competitive, but preferred dimensionality and action heads depend on the evaluation. Its strongest deployment evidence is improved Franka manipulation after latent-action tuning of a VLM backbone; the forward predictor supplies training supervision rather than an inference-time planner.",
    "abstract": "",
    "submittedDate": "2026-08-20",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv260819613",
    "arxivUrl": "https://arxiv.org/abs/2608.19613",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.19613",
    "pdfUrl": "https://arxiv.org/pdf/2608.19613",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "",
    "majorCategory": "VLA",
    "subcategories": [
      "潜动作预训练"
    ],
    "architecture": null,
    "predictionParadigm": null,
    "quadrant": null,
    "classificationStatus": null
  },
  {
    "id": "2608.20430",
    "title": "RISE: Adaptive Imagination for World Action Models",
    "authors": "Lu, Hongbo; Yao, Liang; He, Chenghao; Han, Hao; Liu, Fan; Liao, Wenlong; He, Tao; Peng, Pai",
    "affiliations": "1COWARobot Co. Ltd 2Shanghai Jiao Tong University 3Hohai University",
    "contribution": "We propose RISE (\\textbf{R}efining \\textbf{I}magination through \\textbf{SE}lective Rollout), a system-level adaptive imagination framework that makes sequential \\textsc{Roll}/\\textsc{Stop} decisions according to the expected planning benefit of continued rollout. Experiments on NAVSIM and nuScenes show that RISE achieves the best overall planning performance while reducing unnecessary rollout, with additional transfer results supporting its plug-in generality across WAM architectures.",
    "abstract": "World Action Models (WAMs) improve planning by incorporating future world evolution into action generation, yet existing methods allocate a fixed imagination budget to every scene. We propose RISE (\\textbf{R}efining \\textbf{I}magination through \\textbf{SE}lective Rollout), a system-level adaptive imagination framework that makes sequential \\textsc{Roll}/\\textsc{Stop} decisions according to the expected planning benefit of continued rollout. At each step, a Latent Evaluator estimates the risk revealed by the current prefix and how much planning could improve if imagination continues, while a Rollout Gate weighs this expected benefit against additional computation cost. Since factual driving logs expose only one realized future, we further construct \\textbf{CounterDrive}, a counterfactual dataset with diverse outcomes and risk levels, to enrich future dynamics and provide localized risk supervision. Each retained sample undergoes expert verification and annotation of trajectory validity, incident onset, and causal category, providing a reusable resource for safety-critical world-modeling research. Experiments on NAVSIM and nuScenes show that RISE achieves the best overall planning performance while reducing unnecessary rollout, with additional transfer results supporting its plug-in generality across WAM architectures.",
    "submittedDate": "2026-08-19",
    "primaryCategory": "Latent / Representation WAM",
    "secondaryCategories": [
      "Navigation / Driving / Domain WAM"
    ],
    "bibtexKey": "RISE",
    "arxivUrl": "https://arxiv.org/abs/2608.20430",
    "codeUrls": [
      "https://github.com/COOWAI/RISE"
    ],
    "projectUrl": "https://cowarobot-ai.github.io/RISE/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.20430",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "自动驾驶",
      "高效推理与实时控制",
      "策略后训练与WM-RL"
    ],
    "architecture": "不适用",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.19574",
    "title": "HiTac-WAM: A Hierarchical Tactile World Action Model for Contact-Rich Robot Manipulation",
    "authors": "Xue, Chao; Zhang, Chaofan; Ma, Wenxuan; Yao, Guocai; Cui, Shaowei; Wang, Shuo",
    "affiliations": "1 Institute of Automation, Chinese Academy of Sciences; 2 ImprintX Robotics; 3 Beijing Academy of Artificial Intelligence",
    "contribution": "We present HiTac-WAM, a hierarchical tactile world action model that forecasts a sequence of future tactile states for each candidate action chunk before execution. HiTac-WAM achieves a mean contact F1 of 0.921; under matched training budgets, the directed hierarchy reduces 3D displacement L2 error by 17.6% relative to the deformation-only predictor and improves slip AUPRC by 60.4% relative to the slip-only predictor.",
    "abstract": "World action models jointly predict future visual observations and actions, whereas existing tactile-aware variants typically represent future touch as an image or latent stream without modeling the physical dependencies that organize tactile states hierarchically. We present HiTac-WAM, a hierarchical tactile world action model that forecasts a sequence of future tactile states for each candidate action chunk before execution. The forecast factorizes into contact state, a 3D deformation field, and slip risk, organized as a directed hierarchy in which each downstream stage is conditioned on stop-gradient signals from preceding stages. A directed attention mask allows tactile queries to attend to the video-action context of each candidate while preventing video and action queries from attending to tactile tokens. For planning, HiTac-WAM ranks candidate action chunks using tactile forecasts and task-progress estimates. For execution, the selected tactile forecast is retained as a reference; persistent discrepancies between predicted and observed tactile states trigger corrective replanning. HiTac-WAM achieves a mean contact F1 of 0.921; under matched training budgets, the directed hierarchy reduces 3D displacement L2 error by 17.6% relative to the deformation-only predictor and improves slip AUPRC by 60.4% relative to the slip-only predictor. Across chip grasping, blackboard erasing, and USB insertion, selection guided by the hierarchical forecasts increases the average real-robot success rate from 31.1% to 61.1%, while the full system attains 72.2%.",
    "submittedDate": "2026-08-19",
    "primaryCategory": "Multimodal / Tactile WAM",
    "secondaryCategories": [
      "3D/4D WAM",
      "Latent / Representation WAM"
    ],
    "bibtexKey": "HiTacWAM",
    "arxivUrl": "https://arxiv.org/abs/2608.19574",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.19574",
    "pdfUrl": "https://arxiv.org/pdf/2608.19574",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{HiTacWAM,\n  title={HiTac-WAM: A Hierarchical Tactile World Action Model for Contact-Rich Robot Manipulation},\n  author={Xue, Chao and Zhang, Chaofan and Ma, Wenxuan and Yao, Guocai and Cui, Shaowei and Wang, Shuo},\n  journal={arXiv:2608.19574},\n  year={2026}\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "多模态触觉音频",
      "策略后训练与WM-RL"
    ],
    "architecture": "One Model",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.12854",
    "title": "BrainWAM: Action-Space Coordination of Semantic Priors and Predictive Dynamics for Autonomous Driving",
    "authors": "Zhan, Bing; Shang, Shuyao; Lu, Shuo; Xu, Yuan; Wang, Zhao; Wang, Yida; Zhang, Xueyang; Zhan, Kun; Gu, Jiahao",
    "affiliations": "NLPR, Institute of Automation, Chinese Academy of Sciences (CASIA); Beijing; China; Li Auto Inc.",
    "contribution": "Inspired by neuroscience evidence that complex behavior arises from coordination among functionally specialized systems, we propose BrainWAM, a structured action-space coordination framework that converts semantic reasoning and predictive world modeling into two specialized action-oriented pathways, and aligns them at the level of compact action representations. BrainWAM reaches state-of-the-art performance on both NAVSIM v1 (89.5 PDMS) and NAVSIM v2 (89.6 EPDMS), consistently outperforming VLA-only or WAM-only methods, highlighting BrainWAM as a practical and promising direction for autonomous driving systems.",
    "abstract": "Autonomous driving requires planning under both semantic constraints and predictive dynamics. Existing end-to-end driving approaches, however, typically emphasize only one side of this requirement: Vision-Language-Action (VLA) models exploit VLM priors for semantic reasoning, while World Action Models (WAMs) provide future-aware prediction through generative world modeling. This naturally motivates a unified planner that can leverage both semantic priors and predictive dynamics. However, we find that a naive combination through joint token-level attention suffers from an attention-allocation mismatch, where semantic shortcuts dominate the shared attention space and suppress predictive dynamics. Inspired by neuroscience evidence that complex behavior arises from coordination among functionally specialized systems, we propose BrainWAM, a structured action-space coordination framework that converts semantic reasoning and predictive world modeling into two specialized action-oriented pathways, and aligns them at the level of compact action representations. We further introduce an asynchronous rectified-flow inference strategy with decoupled video and action denoising, which shortens inference latency while preserving planning-relevant predictive context. BrainWAM reaches state-of-the-art performance on both NAVSIM v1 (89.5 PDMS) and NAVSIM v2 (89.6 EPDMS), consistently outperforming VLA-only or WAM-only methods, highlighting BrainWAM as a practical and promising direction for autonomous driving systems.",
    "submittedDate": "2026-08-19",
    "primaryCategory": "Navigation / Driving / Domain WAM",
    "secondaryCategories": [
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "BrainWAM",
    "arxivUrl": "https://arxiv.org/abs/2608.12854",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.12854",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "自动驾驶",
      "联合视频动作建模",
      "高效推理与实时控制"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "联合预测",
    "quadrant": "Q3 · Dual-system × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.07267",
    "title": "WNM-3D: A World Navigation Model with 3D Scene Conditioning for Closed-Loop VLN",
    "authors": "Huang, Yuehao; Wu, Yunzi; Zhang, Xiaotao; Li, Xinhai; Dong, Jiankun; Lv, Jiajun; Zhang, Chi; Bai, Chenjia; Liu, Yong; Li, Xuelong",
    "affiliations": "1Institute of Artificial Intelligence, China Telecom,2Zhejiang University,3Tongji University; 4Shanghai Jiao Tong University",
    "contribution": "We present WNM-3D, a generative World Navigation Model with 3D scene conditioning for continuous VLN. Experiments on GN-Bench show that WNM-3D outperforms strong VLM-based navigation policies and its 2D-conditioned counterpart in closed-loop navigation.",
    "abstract": "Recent vision-language navigation (VLN) systems increasingly adapt pretrained vision-language models (VLMs) into vision-language-action (VLA) policies that map egocentric observations and language instructions directly to navigation actions. Although semantically capable, such action-centric training does not explicitly model how the agent's visual observations should evolve under its predicted motion. Generative world-action models (WAMs) jointly predict future observations and actions, yet existing WAMs for continuous VLN do not condition joint future-view and action generation on geometry-aware representations inferred from the observed history. We present WNM-3D, a generative World Navigation Model with 3D scene conditioning for continuous VLN. To consolidate past observations into persistent scene context, a frozen feed-forward geometry encoder extracts geometry-aware representations from the monocular egocentric RGB history, and a trainable 3D Scene-to-Token Adapter converts them into a fixed-length prefix in the token space of the world-action Diffusion Transformer. Through block-causal attention, this prefix conditions every future video-action block, providing a shared geometric context for both future-view and action generation. We train WNM-3D through supervised world-action fine-tuning on A*-generated demonstrations, DAgger-style adaptation on policy-visited states, and Counterfactual DanceGRPO refinement for closed-loop execution. Experiments on GN-Bench show that WNM-3D outperforms strong VLM-based navigation policies and its 2D-conditioned counterpart in closed-loop navigation. Stage-wise ablations further show that DAgger-SFT provides the larger success-rate gain, while Counterfactual DanceGRPO subsequently improves both navigation success and path efficiency.",
    "submittedDate": "2026-08-19",
    "primaryCategory": "3D/4D WAM",
    "secondaryCategories": [
      "Navigation / Driving / Domain WAM",
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "WNM3D",
    "arxivUrl": "https://arxiv.org/abs/2608.07267",
    "codeUrls": [
      "https://github.com/TeleHuman/WNM-3D"
    ],
    "projectUrl": "https://wnm-3d.github.io/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.07267",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "导航",
      "三维多视角建模",
      "联合视频动作建模"
    ],
    "architecture": "One Model",
    "predictionParadigm": "联合预测",
    "quadrant": "Q1 · One Model × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.18077",
    "title": "Hydra-0: Action Flow for Generalist World Modeling and Control",
    "authors": "Li, Hongyu; Wen, Bowen; Zhu, Xinghao; Wang, Yixuan; Du, Yilun; Li, Yunzhu; Konidaris, George; Birchfield, Stan; Pouya, Soha; Li, Chenran; Chang, Yan",
    "affiliations": "NVIDIA Brown University Columbia University Harvard University",
    "contribution": "We introduce Hydra-0, a generalist world model conditioned on action flow, which represents robot actions as pixel motion. Our best configuration achieves 90.4% lower robot-motion error and 60.2% lower object-motion error than our action-conditioned baseline, while supporting zero-shot composition and data-efficient adaptation.",
    "abstract": "We introduce Hydra-0, a generalist world model conditioned on action flow, which represents robot actions as pixel motion. This shared visual interface enables generalist world modeling and control by learning action consequences across embodiments, tasks, environments, and video-generation backbones. Our best configuration achieves 90.4% lower robot-motion error and 60.2% lower object-motion error than our action-conditioned baseline, while supporting zero-shot composition and data-efficient adaptation. On the RoboLab benchmark, Hydra-0 achieves a Pearson correlation of r=0.96 between replayed and reference success rates. Finally, we uncover an emergent inverse mode of this interface: a world action model that predicts compatible robot motion from desired object flow transferred from a human demonstration. A trained action head maps the resulting latent features to executable actions without requiring task-specific expert robot demonstrations. Together, these results demonstrate the potential of action flow as a shared control interface connecting heterogeneous training data, open-loop policy evaluation, and robot control.",
    "submittedDate": "2026-08-18",
    "primaryCategory": "Latent / Representation WAM",
    "secondaryCategories": [
      "Evaluation / Survey / Theory",
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "Hydra0",
    "arxivUrl": "https://arxiv.org/abs/2608.18077",
    "codeUrls": [],
    "projectUrl": "https://nvidia-isaac.github.io/video_to_data/hydra-0/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.18077",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "视觉规划与IDM",
      "泛化与动作对齐"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.17209",
    "title": "Teach and Grow: An Agent-Centered Architecture for General Robot Learning",
    "authors": "Nie, Chang; Liu, Zhe; Wang, Hesheng",
    "affiliations": "Chang Nie, Zhe Liu, and Hesheng Wang are with the School of Automation and Intelligent Sensing, Shanghai Jiao Tong University, and the Shanghai Key Laboratory of Navigation and Location Based Services, Shanghai 200240, China. Corresponding author: Hesheng Wang (e-mail: )",
    "contribution": "We present Teach-and-Grow Learning (TGL), an agent-centered architecture for general robot learning. Our LIBERO evaluation attains state-of-the-art performance; controlled studies expose skill induction, persistent reuse, and agent-directed adaptation.",
    "abstract": "End-to-end vision-language-action (VLA) and world-action models offer an elegant route to general-purpose robotics, but their reliability is bounded by validated physical coverage. When an unfamiliar object, sensor, embodiment, or contact falls outside that coverage and no validated fallback exists, correcting the failure requires new robot data, a policy update, and regression testing. This recurring burden is the retraining tax. Unlike text, embodied data must often be created by operating machines. We present Teach-and-Grow Learning (TGL), an agent-centered architecture for general robot learning. In its general form, a multimodal agent turns a few successful demonstrations into reusable Skill Blocks: closed-loop behaviors for meaningful subgoals. In a new scene, the agent grounds and composes these blocks, selects learned or geometric tools, observes the physical outcome, and revises the route when execution departs from intent. A Skill Library stores executable behavior, while structured Experience Memory carries forward success, failure, and repair. New tasks are acquired without task-specific policy retraining. Our LIBERO evaluation attains state-of-the-art performance; controlled studies expose skill induction, persistent reuse, and agent-directed adaptation. Finally, we propose the Teach-and-Grow scaling-law hypothesis: if X denotes effective reusable experience, future-task error and teaching demand should approach irreducible floors as power laws in X. The architecture therefore treats deployment as a period of continued learning, in which one task can make the next easier.",
    "submittedDate": "2026-08-17",
    "primaryCategory": "3D/4D WAM",
    "secondaryCategories": [
      "Evaluation / Survey / Theory",
      "Memory WAM",
      "Multimodal / Tactile WAM"
    ],
    "bibtexKey": "TGL",
    "arxivUrl": "https://arxiv.org/abs/2608.17209",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.17209",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "Related resources",
    "subcategories": [
      "理论与规划",
      "综述与技术资源"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.10780",
    "title": "StageWAM: Joint-Embedding Stage Prediction for World-Action Models in Robot Manipulation",
    "authors": "Liu, Xiao; Yang, Yuguang; Wang, Xi; Jiang, Kai; Chi, Cheng; Xu, Yong; Ding, Wenchao; Chen, Yilun; Wang, Yan",
    "affiliations": "1Institute for AI Industry Research (AIR), Tsinghua University; 2School of Electronic Information Engineering, Beihang University; 3AIR Wuxi Innovation Center, Tsinghua University; 4School of Artificial Intelligence, Beihang University; 5School of Information, Renmin University of China; 6TARS Robotics",
    "contribution": "We introduce StageWAM, which augments a Motus-based World Action Model (WAM) with Stage-JEPA, a goal-conditioned Joint-Embedding Predictive Architecture (JEPA) predictor. Across 50 RoboTwin 2.0 tasks in clean and randomized environments, StageWAM achieves 90.25% overall success and reduces the mean number of execution steps in successful rollouts by 5.97% relative to the strongest baseline.",
    "abstract": "Generalist robot policies aim to map multimodal observations and linguistic task instructions to actions across diverse tasks. However, existing methods typically represent the future as a fixed, short video-action chunk. This short-term future captures local scene evolution for action execution, but it does not explicitly describe the stage-level future that specifies how a task should progress from its current stage to the next. We therefore distinguish two complementary futures for robot manipulation: a short-term physical future to capture local scene evolution and a stage-level semantic future to represent task progress. We introduce StageWAM, which augments a Motus-based World Action Model (WAM) with Stage-JEPA, a goal-conditioned Joint-Embedding Predictive Architecture (JEPA) predictor. Given the current observation and task instruction, Stage-JEPA uses a frozen V-JEPA2 encoder to extract the current-state representation and predicts the latent target of the next inferred stage. Across 50 RoboTwin 2.0 tasks in clean and randomized environments, StageWAM achieves 90.25% overall success and reduces the mean number of execution steps in successful rollouts by 5.97% relative to the strongest baseline.",
    "submittedDate": "2026-08-14",
    "primaryCategory": "Latent / Representation WAM",
    "secondaryCategories": [
      "Multimodal / Tactile WAM"
    ],
    "bibtexKey": "StageWAM",
    "arxivUrl": "https://arxiv.org/abs/2608.10780",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.10780",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "潜空间预测与JEPA",
      "联合视频动作建模",
      "记忆与长时序"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "联合预测",
    "quadrant": "Q3 · Dual-system × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.03682",
    "title": "PhyAI: Real-Time Physical AI at the Edge, Scalable Rollouts in the Cloud",
    "authors": "Chenghua Wang; Daliang Xu; Dongqi Cai; Duojin Sun; Hao Zhang; Haoze Qian; Huaiyuan Zhang; Jinshuo Cui; Junbo Cui; Kezhao Zhao; Longxi Gao; Mengwei Xu; Rongjie Yi; Ruixin Liu; Shangguang Wang; Tam Sikyuen; Tianyue Zhang; Weikai Xie; Xuanzhe Liu; Yingying Qin; Yiwen Lu; Yuan Yao; Yuezhi Zu; Yunhan Guo; Yuxin Zheng; Ziqi Guo",
    "affiliations": "1Beijing University of Posts and Telecommunications 2Nanjing University 3Peking University; 4Tsinghua University 5MingTi Technology 6ModelBest",
    "contribution": "To unify them, we build PhyAI, a Physical AI inference engine with a single runtime that keeps architecture-specific conditioning, solver, cache, and output logic in model adapters while sharing graph execution, kernels, memory management, and parallel services. PhyAI achieves 1.40x-4.65x speedups over the official implementations of pi0, pi0.5, GR00T N1.7, and MiniCPM-Robot.",
    "abstract": "Physical AI policies require inference throughout their lifecycle, including model evaluation, cloud reinforcement learning rollout, edge GPU serving, and onboard deployment. Although these settings share the same checkpoint and action semantics, they often rely on separate inference programs. To unify them, we build PhyAI, a Physical AI inference engine with a single runtime that keeps architecture-specific conditioning, solver, cache, and output logic in model adapters while sharing graph execution, kernels, memory management, and parallel services. The same codebase runs vision-language-action (VLA) models and world-action models (WAMs) on single or multiple GPUs across onboard, edge, and cloud deployments. We used the adapter interface to add MiniCPM-Robot on the day of its release. PhyAI achieves 1.40x-4.65x speedups over the official implementations of pi0, pi0.5, GR00T N1.7, and MiniCPM-Robot. On Cosmos3-Nano-Policy-DROID it reduces latency from 2.46 to 1.18 s on eight H20 GPUs (CFG=2, TP=4), a 2.08x speedup. Specialized runtimes remain faster in several configurations, so our goal is one runtime with competitive latency rather than the fastest result in every case. Detailed profiles reveal why different models need different execution policies: on a Hopper-series GPU at batch size one, the pi0.5 action expert accounts for 8.8% of FLOPs but 57.2% of latency; at batch size 32 its share drops to 13.5% and throughput reaches about 100 samples/s. Cosmos3 remains generation-dominated and gains only 14.3% throughput as batch size increases from 1 to 16. We further introduce the control-time Roofline, which distinguishes inference-bound from environment-bound control; the measured pi0.5 points on four LIBERO suites are environment-bound while Cosmos3 stays inference-bound. Code and benchmarks: https://github.com/mingti-org/phyai.",
    "submittedDate": "2026-08-14",
    "primaryCategory": "Real-Time / Efficient WAM",
    "secondaryCategories": [
      "Evaluation / Survey / Theory",
      "Memory WAM",
      "WAM + RL"
    ],
    "bibtexKey": "PhyAI",
    "arxivUrl": "https://arxiv.org/abs/2608.03682",
    "codeUrls": [
      "https://github.com/mingti-org/phyai"
    ],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.03682",
    "pdfUrl": "https://arxiv.org/pdf/2608.03682",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{PhyAI,\n  title={PhyAI: Real-Time Physical AI at the Edge, Scalable Rollouts in the Cloud},\n  author={Chenghua Wang and Daliang Xu and Dongqi Cai and Duojin Sun and Hao Zhang and Haoze Qian and Huaiyuan Zhang and Jinshuo Cui and Junbo Cui and Kezhao Zhao and Longxi Gao and Mengwei Xu and Rongjie Yi and Ruixin Liu and Shangguang Wang and Tam Sikyuen and Tianyue Zhang and Weikai Xie and Xuanzhe Liu and Yingying Qin and Yiwen Lu and Yuan Yao and Yuezhi Zu and Yunhan Guo and Yuxin Zheng and Ziqi Guo},\n  journal={arXiv:2608.03682},\n  year={2026}\n}",
    "majorCategory": "Related resources",
    "subcategories": [
      "综述与技术资源",
      "训练优化与蒸馏"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.12416",
    "title": "RoboSynChallenge: Mastering Real-World Dexterity via Generalizing Synthesized Manipulation Skills",
    "authors": "Zhao, Runyi; Wu, Ruixin; Li, Chengkun; Zhang, Hongrui; Li, Ang; Jin, Ruixing; Deng, Yueci; Guo, Yingying; Ding, Lihe; Dong, Shaocong; Xue, Tianfan; Gao, Yanjun; Luo, Yudong; Poupart, Pascal; Wu, Simo; Jia, Kui; Zheng, Wei-shi; Liu, Guiliang",
    "affiliations": "1 Shenzhen Loop Area Institute (SLAI) 2 The Chinese University of Hong Kong, Shenzhen; 3 The Chinese University of Hong Kong; 4 The Hong Kong University of Science and Technology; 5 LARK Lab, University of Colorado Anschutz; 6 Mila - Quebec AI Institute, Canada; 7 Vector Institute; 8 University of Waterloo; 9 Fudan University; 10 Sun Yat-sen University",
    "contribution": "Achieving generalizable robotic manipulation remains a central challenge in embodied intelligence. Despite rapid advances in model architectures and learning algorithms, progress is often limited by the scarcity and narrow diversity of real-world data.",
    "abstract": "Achieving generalizable robotic manipulation remains a central challenge in embodied intelligence. Despite rapid advances in model architectures and learning algorithms, progress is often limited by the scarcity and narrow diversity of real-world data. The RoboSynChallenge competition introduces a unified benchmark to evaluate and advance the generalizability of manipulation policies across a spectrum of tasks, environments, and difficulty levels. To alleviate the shortage of realistic data, the challenge integrates large-scale synthetic data generation with standardized real-world robotic evaluation. Participants are encouraged to leverage synthesized state-action trials to improve general-purpose policy learning, while final assessments are conducted exclusively on unseen real-world manipulation environments. Baseline implementations, including Transformer-, Diffusion-, Vision-Language-Action, and World-Action-Model-based policies, are provided to ensure reproducibility and comparability. By coupling scalable simulation-based training with rigorous real-world validation, RoboSynChallenge aims to foster the development of broadly capable, data-efficient, and adaptable manipulation systems, thereby paving the way toward truly general robotic intelligence.",
    "submittedDate": "2026-08-12",
    "primaryCategory": "Real-Time / Efficient WAM",
    "secondaryCategories": [
      "Evaluation / Survey / Theory"
    ],
    "bibtexKey": "RoboSynChallenge",
    "arxivUrl": "https://arxiv.org/abs/2608.12416",
    "codeUrls": [
      "https://github.com/EDEM-AI/RoboSynChallenge"
    ],
    "projectUrl": "https://robosyn-bench.net/",
    "venue": "NeurIPS 2026 Competition Track",
    "paperUrl": "https://arxiv.org/abs/2608.12416",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "机器人操作基准",
      "仿真到真实评测",
      "合成数据与数据生成"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.11521",
    "title": "Keep the Future, Drop the Rollout: RIFT for World Action Models",
    "authors": "Zhang, Chushan; Tong, Jinguang; Li, Xuesong; Wang, Yikai; Li, Hongdong",
    "affiliations": "1Australian National University 2Beijing Normal University",
    "contribution": "On LIBERO, RIFT achieves 98.8%98.8\\% success, close to rollout-based Joint, IDM, and LingBot-VA at 98.4%98.4\\% to 98.6%98.6\\%, while reducing action-chunk latency by 68.2%68.2\\% to 89.1%89.1\\%.",
    "abstract": "World action models (WAMs) condition robot actions on predicted futures, but iterative video rollout increases deployment latency. We ask whether action generation requires the evolving rollout trajectory or only its future representation. Across four WAMs on all 40 LIBERO tasks, paired closed-loop interventions show that masking or reassigning future-cache values changes execution and reduces success, indicating sensitivity to future values and their assigned positions. For Joint and Cosmos-2, however, replaying one fixed final-clean key/value (K/V) cache nearly preserves unmodified execution, with 1.71.7 to 1.91.9~cm end-effector average displacement error and 97.9%97.9\\% to 98.2%98.2\\% success. This separates cache consumption from production: these models can reuse a fixed cache but still require iterative rollout to construct it. We therefore propose RIFT (\\emph{Rollout-free Imagination via Future Tokens}), which uses learned anticipation tokens to construct a complete future K/V cache in one backbone pass while retaining the original future-read interface. On LIBERO, RIFT achieves 98.8%98.8\\% success, close to rollout-based Joint, IDM, and LingBot-VA at 98.4%98.4\\% to 98.6%98.6\\%, while reducing action-chunk latency by 68.2%68.2\\% to 89.1%89.1\\%. On RoboTwin~2.0, RIFT reaches 92.9/92.6%92.9/92.6\\% on clean/randomized scenes, the highest observed among the evaluated methods. These results support rollout-free future conditioning without iterative video generation at deployment.",
    "submittedDate": "2026-08-12",
    "primaryCategory": "Real-Time / Efficient WAM",
    "secondaryCategories": [],
    "bibtexKey": "RIFT",
    "arxivUrl": "https://arxiv.org/abs/2608.11521",
    "codeUrls": [
      "https://github.com/ChushanZhang/RIFT"
    ],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.11521",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "潜空间预测与JEPA",
      "视觉规划与IDM",
      "高效推理与实时控制"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.08023",
    "title": "4D-WAM: Infusing Spatiotemporal Awareness into World Action Models through Trajectory Fields",
    "authors": "Yang, Lishan; Song, Wenxuan; Wang, Xi; Sheng, Pingyue; Fang, Zheng; Zhou, Ziyang; He, Junjie; Yan, Haodong; Chen, Jiayi; Sun, Nan; Sun, Qiao; Wang, Pengwei; Liu, Lingqiao; Wang, Yan; Gao, Yuxiang; Dayoub, Feras; Li, Haoang",
    "affiliations": "1Adelaide University,2HKUST(GZ), 3COCO Matrix,4Tsinghua University,5Fudan University,6BAAI",
    "contribution": "In this work, we propose 4D-WAM, a model-agnostic training strategy that injects spatiotemporal knowledge from 3D trajectory fields into WAMs through representation alignment. To this end, we introduce two complementary objectives: 1) motion alignment, which aligns temporal feature variations across adjacent frames and encourages the model to build local 4D awareness during training, and 2) destination alignment, which guides the model to infer the final destination from the source frame by minimizing the gap between their attention-like similarity distributions.",
    "abstract": "Building on recent advances in world models, World Action Models (WAMs) jointly model video prediction and action generation. However, they typically represent videos in 2D pixel space, creating a representation gap with 3D space in which robotic actions are executed. Recent 3D approaches introduce 3D information, but fail to fully exploit the dynamics of 3D structures. In this work, we propose 4D-WAM, a model-agnostic training strategy that injects spatiotemporal knowledge from 3D trajectory fields into WAMs through representation alignment. To this end, we introduce two complementary objectives: 1) motion alignment, which aligns temporal feature variations across adjacent frames and encourages the model to build local 4D awareness during training, and 2) destination alignment, which guides the model to infer the final destination from the source frame by minimizing the gap between their attention-like similarity distributions. Together, these objectives provide both local motion supervision and long-horizon goal guidance, enabling WAMs to learn trajectory-level spatiotemporal representations. Extensive in-distribution and out-of-distribution experiments across different base models demonstrate the model's improvements in spatial understanding, execution precision, robustness, generalization, and versatility.",
    "submittedDate": "2026-08-12",
    "primaryCategory": "3D/4D WAM",
    "secondaryCategories": [],
    "bibtexKey": "4DWAM",
    "arxivUrl": "https://arxiv.org/abs/2608.08023",
    "codeUrls": [
      "https://github.com/lishanyqy/4DWAM"
    ],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.08023",
    "pdfUrl": "https://arxiv.org/pdf/2608.08023",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{4DWAM,\n  title={4D-WAM: Infusing Spatiotemporal Awareness into World Action Models through Trajectory Fields},\n  author={Yang, Lishan and Song, Wenxuan and Wang, Xi and Sheng, Pingyue and Fang, Zheng and Zhou, Ziyang and He, Junjie and Yan, Haodong and Chen, Jiayi and Sun, Nan and Sun, Qiao and Wang, Pengwei and Liu, Lingqiao and Wang, Yan and Gao, Yuxiang and Dayoub, Feras and Li, Haoang},\n  journal={arXiv:2608.08023},\n  year={2026}\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "三维多视角建模",
      "泛化与动作对齐"
    ],
    "architecture": "不适用",
    "predictionParadigm": "联合预测",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.11605",
    "title": "Foresight Without Seeing: Latent Futures for World Action Models",
    "authors": "Huang, Jiakai; Wu, Zhongbo; Zhang, Zheng; Wang, Zihan; You, Shan; Huang, Tao",
    "affiliations": "1 Shanghai Jiao Tong University; 2 ACE Robotics; 3 Nanyang Technological University",
    "contribution": "To bridge this gap, we propose ForeWAM, a dynamics-conditioned direct-policy WAM that provides predictive context for action generation without decoding future videos. Without embodied robot data pretraining, the standard and accelerated variants of ForeWAM achieve average success rates of 96.7% and 96.9% on LIBERO, respectively.",
    "abstract": "World Action Models (WAMs) couple future visual prediction with robot action generation, enabling policies to model how the physical world evolves during interaction. Existing WAMs differ in how predictive dynamics are exposed to the action pathway. Explicit-future WAMs provide direct access to predicted scene evolution, but incur substantial inference costs from iterative video denoising. In contrast, direct-policy WAMs efficiently predict actions from the current observation but lack an explicit inference-time interface for exposing predictive dynamics to the Action DiT. To bridge this gap, we propose ForeWAM, a dynamics-conditioned direct-policy WAM that provides predictive context for action generation without decoding future videos. At its core, Future-KV performs a single Video DiT prefill over the current visual latent and stochastic future slots, and reuses the resulting layer-wise key-value states throughout action denoising. We further introduce dynamics registers supervised by a frozen latent action teacher, encouraging the implicit future states to capture interaction-induced transitions such as object motion, contact changes, and task progress. Ground-truth future observations and the teacher are used only during training; deployment requires neither and performs no future video generation. Without embodied robot data pretraining, the standard and accelerated variants of ForeWAM achieve average success rates of 96.7% and 96.9% on LIBERO, respectively. The standard variant further achieves 61.6% success on LIBERO-Plus. These results demonstrate that direct-policy WAMs can retain efficient action prediction while exposing predictive dynamics to the action pathway without explicitly generating future observations.",
    "submittedDate": "2026-08-11",
    "primaryCategory": "Latent / Representation WAM",
    "secondaryCategories": [
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "ForeWAM",
    "arxivUrl": "https://arxiv.org/abs/2608.11605",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.11605",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "潜空间预测与JEPA",
      "视觉规划与IDM",
      "高效推理与实时控制"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.11204",
    "title": "Surgical WAM: A World-Action Model for Data-Efficient Surgical Robot Learning",
    "authors": "Bao, Wenrui; Jiang, Tianyun; Chen, Zhiben; Lim, Ser-Nam; Peng, Peter D.; Shang, Yuzhang",
    "affiliations": "Not identified",
    "contribution": "To answer it, we introduce the Surgical World-Action Model (Surgical WAM), a unified generative model built on Cosmos Policy that jointly predicts future endoscopic observations and executable surgical robot action chunks.",
    "abstract": "Learning reliable surgical manipulation policies is bottlenecked by the scarcity of action-labeled demonstrations: teleoperated surgical robot (e.g., dVRK) trajectories with synchronized kinematics are costly to collect, while surgical tasks demand precise contact handling, long-horizon reasoning, and bimanual coordination. Endoscopic video is comparatively inexpensive and abundant relative to synchronized video--kinematics trajectories, and a natural way to exploit it is to learn world models of surgical scenes. However, existing surgical world models use video primarily for simulation or policy evaluation, and rarely translate the learned dynamics into closed-loop control. This gap raises our central question: under a fixed budget of action-labeled demonstrations, does action-free video pretraining improve closed-loop surgical manipulation? To answer it, we introduce the Surgical World-Action Model (Surgical WAM), a unified generative model built on Cosmos Policy that jointly predicts future endoscopic observations and executable surgical robot action chunks. Surgical WAM first learns surgical visual dynamics from action-free video and is then fine-tuned on the fixed action-labeled budget; at deployment, it acts as a closed-loop, receding-horizon controller that executes a short prefix of each predicted action chunk and replans from the resulting observation. On a suite of four simulated surgical manipulation tasks, video pretraining improves the average success rate from 63.5% to 77.8%, including an absolute gain of 20 percentage points on PegTransfer, with the largest improvements on contact-rich and bimanual tasks. These results demonstrate that action-free video provides transferable visual dynamics priors for learning surgical robot control with limited action supervision, positioning data-efficient video pretraining as a practical path toward scaling up surgical robot learning.",
    "submittedDate": "2026-08-11",
    "primaryCategory": "Real-Time / Efficient WAM",
    "secondaryCategories": [
      "Evaluation / Survey / Theory",
      "Navigation / Driving / Domain WAM"
    ],
    "bibtexKey": "SurgicalWAM",
    "arxivUrl": "https://arxiv.org/abs/2608.11204",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.11204",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "联合视频动作建模",
      "泛化与动作对齐"
    ],
    "architecture": "One Model",
    "predictionParadigm": "联合预测",
    "quadrant": "Q1 · One Model × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.01880",
    "title": "World Action Models in Real Time: An Empirical Study of Smooth Execution via Asynchronous Deployment",
    "authors": "Team, Motubrain",
    "affiliations": "Not identified",
    "contribution": "We present an empirical study of asynchronous deployment strategies that overlap model inference with action execution to enable responsive and smooth control. In contrast, prefix-conditioned generation achieves the best overall balance between task performance, execution speed, and trajectory smoothness by learning consistent action continuations during training.",
    "abstract": "World Action Models generate fixed-horizon action chunks through iterative denoising, creating substantial inference latency that can cause pauses, stale actions, and discontinuities during robotic execution. We present an empirical study of asynchronous deployment strategies that overlap model inference with action execution to enable responsive and smooth control. We compare six strategies, including synchronous execution, pure asynchronous switching, post-hoc action blending, denoising-time blending, inference-time velocity guidance, and prefix-conditioned generation, on a 10 Hz bimanual robot. Evaluation combines offline trajectory analysis with online experiments across dynamic manipulation, precision-critical placement, and long-horizon tasks. Our results identify accurate temporal alignment between observations, predictions, and executed commands as a fundamental requirement. Alignment errors produce persistent chunk-boundary discontinuities that cannot be corrected through blending alone. With proper alignment, direct action weighting provides a simple and smooth baseline but sacrifices accuracy in precision-critical tasks. Inference-time velocity guidance fails to reliably constrain committed actions on our platform. In contrast, prefix-conditioned generation achieves the best overall balance between task performance, execution speed, and trajectory smoothness by learning consistent action continuations during training. These findings clarify the practical trade-offs among asynchronous deployment strategies and provide guidance for deploying high-latency World Action Models in real-time robotic systems.",
    "submittedDate": "2026-08-11",
    "primaryCategory": "Real-Time / Efficient WAM",
    "secondaryCategories": [
      "Evaluation / Survey / Theory"
    ],
    "bibtexKey": "RTESSEAD",
    "arxivUrl": "https://arxiv.org/abs/2608.01880",
    "codeUrls": [],
    "projectUrl": "https://www.motubrain.com/en/research/beyond-stalls-deploying-world-action-models/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.01880",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "Related resources",
    "subcategories": [
      "评估指标与协议",
      "综述与技术资源"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.10232",
    "title": "FACT: Failure-Aware Causal Training for World-Action Models",
    "authors": "Peng, Quanquan; Liang, Yutong; Yan, Rui; Hansen, Nicklas; Wang, Xiaolong",
    "affiliations": "University of California San Diego",
    "contribution": "We introduce FACT, a causal World-Action Model that predicts future video and task progress conditioned on the executed action. Extensive experiments on simulation and real-world bimanual manipulation tasks show that FACT outperforms many existing baselines, improves as failure data are incorporated into training, and reduces success-biased future hallucination under bad actions.",
    "abstract": "Recent world-action models (WAMs) show that co-training policies with future prediction can provide physical priors for action generation. Building on the future-prediction ability of video models, many WAMs generate future videos and recover actions with inverse-dynamics models, or use these predicted videos as goal conditions for action generation. In both cases, the world model is trained mostly on successful demonstrations and has little reason to predict the consequences of bad actions. We introduce FACT, a causal World-Action Model that predicts future video and task progress conditioned on the executed action. This action-conditioned interface allows failure rollouts to supervise action consequences, turning bad actions into valid future targets rather than being discarded. Failure-aware training makes the progress predictor aware of both successful and failed action outcomes, which can optionally be used to score sampled action candidates at inference. Extensive experiments on simulation and real-world bimanual manipulation tasks show that FACT outperforms many existing baselines, improves as failure data are incorporated into training, and reduces success-biased future hallucination under bad actions. See more details at https://fact-wam.github.io/",
    "submittedDate": "2026-08-10",
    "primaryCategory": "General WAM",
    "secondaryCategories": [],
    "bibtexKey": "FACT",
    "arxivUrl": "https://arxiv.org/abs/2608.10232",
    "codeUrls": [
      "https://github.com/Bariona/FACT"
    ],
    "projectUrl": "https://fact-wam.github.io/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.10232",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "策略后训练与WM-RL",
      "联合视频动作建模"
    ],
    "architecture": "One Model",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.10107",
    "title": "4D-WAM: 4D Consistent World Modeling for Autonomous Driving",
    "authors": "Fu, Jiacheng; Yuan, Yibo; Tian, Meng; Li, Yue; Zhu, Jiangtong; Han, Jianhua; Zhang, Yueyi; Fang, Jianwu; Xue, Jianru; Xu, Hang; Xiong, Zhiwei",
    "affiliations": "1University of Science and Technology of China2Xi’an Jiaotong University; 3Yinwang Intelligent Technology Co., Ltd.4Midea Group",
    "contribution": "To alleviate this issue, we present 4D-WAM, a model that leverages geometric foundation models for training-time supervision to enable 4D consistent world modeling. Extensive experiments demonstrate that 4D-WAM effectively models 4D consistent scene evolution and achieves state-of-the-art performance on challenging NAVSIM-v1 and NAVSIM-v2 benchmarks.",
    "abstract": "Emerging World-Action Models (WAMs) have demonstrated promising performance in autonomous driving by jointly modeling future driving scene evolution and trajectory planning. However, existing WAMs are typically trained with video data, which is only 2D projections of the underlying 4D driving scene. Consequently, WAMs fail to understand and capture the structure of 4D scenes and thus generate visually plausible yet 4D inconsistent future predictions that mislead downstream planning. To alleviate this issue, we present 4D-WAM, a model that leverages geometric foundation models for training-time supervision to enable 4D consistent world modeling. Specifically, we feed WAM-predicted future frames into a geometric foundation model, and use 4D-aware responses to define a 4D consistency loss. This loss encourages the model to understand, represent, and predict physically consistent 4D scenes during training, without additional inference cost. Moreover, we identify an early-decision phenomenon in WAMs and propose a decision-oriented timestep sampling strategy that emphasizes supervision at early, high-noise stages, where driving decisions are primarily formed. By propagating 4D supervision to this critical decision-formation phase, the proposed strategy further improves trajectory planning. Extensive experiments demonstrate that 4D-WAM effectively models 4D consistent scene evolution and achieves state-of-the-art performance on challenging NAVSIM-v1 and NAVSIM-v2 benchmarks.",
    "submittedDate": "2026-08-10",
    "primaryCategory": "3D/4D WAM",
    "secondaryCategories": [
      "Navigation / Driving / Domain WAM"
    ],
    "bibtexKey": "4DWAM2",
    "arxivUrl": "https://arxiv.org/abs/2608.10107",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.10107",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "三维多视角建模",
      "自动驾驶"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "联合预测",
    "quadrant": "Q3 · Dual-system × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.09771",
    "title": "SLIM-0.5B: Learning Action-Grounded Predictive Latents for Robot Manipulation",
    "authors": "Wang, Jingkai; Tang, Zihan; Zhang, Gu; Cao, Mingyu; Chen, Jiapeng; Zhao, Jingjiao; Chen, Xiansheng; Wang, Pengwei; Liu, Lemao; Dou, Dejing",
    "affiliations": "1 Fudan University; 2 Beijing Academy of Artificial Intelligence; 3 Tsinghua University; 4 Renmin University of China",
    "contribution": "We propose SLIM (Self-supervised Latent Interaction Model), a compact 0.5B-parameter latent interaction policy.",
    "abstract": "Vision-language-action policies rely on large multimodal backbones to jointly perform perception, language conditioning, and action generation at every control step. Much of this capacity supports open-domain semantics, whereas continuous robot manipulation primarily requires compact representations of observations, actions, and the transitions induced by actions. Pixel-level world models provide another route, but predicting visual details irrelevant to control can be unnecessarily expensive. We propose SLIM (Self-supervised Latent Interaction Model), a compact 0.5B-parameter latent interaction policy. SLIM learns action-grounded predictive latents that capture both action-conditioned future transitions and the actions that explain observed changes. SLIM learns these representations through self-supervised masked trajectory prediction, combining action reconstruction with future-latent prediction. A compact Mixture-of-Transformers (MoT) backbone models interactions between observation latents and action tokens. The resulting policy is trained with flow matching for language-conditioned action generation. Across simulation benchmarks and real-world evaluation, SLIM matches or exceeds representative large-scale VLA and world-action-model baselines with fewer parameters, no additional embodied pretraining, lower inference latency, and substantially lower GPU memory usage.",
    "submittedDate": "2026-08-10",
    "primaryCategory": "Memory WAM",
    "secondaryCategories": [
      "Evaluation / Survey / Theory",
      "Latent / Representation WAM",
      "Multimodal / Tactile WAM",
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "SLIM05B",
    "arxivUrl": "https://arxiv.org/abs/2608.09771",
    "codeUrls": [
      "https://github.com/kzz1031/SLIM"
    ],
    "projectUrl": "https://kzz1031.github.io/slim-project-page/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.09771",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "VLA",
    "subcategories": [
      "未来表征辅助VLA",
      "扩散与流匹配VLA"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.09730",
    "title": "World Tokens: Enhancing Embodied Policies with Training-Time World Modeling",
    "authors": "Tang, Qu; Zhuang, Benhui; Yuan, Bo; Yu, Xue; Guo, Longteng; Feng, Junlan",
    "affiliations": "JIUTIAN Research Zhongguancun Academy",
    "contribution": "We introduce World Tokens, an embodied policy architecture built around a World Adapter that bridges visual-language understanding, world-dynamics modeling, and action generation.",
    "abstract": "Vision-language-action (VLA) models are a widely adopted paradigm for embodied policies. They excel at efficient closed-loop control but do not explicitly model how physical scenes evolve as a task unfolds. Recently emerging world-action models (WAMs) leverage pretrained video world models to capture spatiotemporal evolution, yet retaining future generation or a large video backbone in the control loop substantially increases inference cost. We introduce World Tokens, an embodied policy architecture built around a World Adapter that bridges visual-language understanding, world-dynamics modeling, and action generation. It uses world modeling during training to enhance the action policy while preserving efficient deployment. Specifically, the World Adapter transforms VLM features into a fixed set of world tokens, which condition a jointly fine-tuned future-video denoiser and simultaneously serve as the action expert's sole visual-language context. This shared conditioning allows gradients from future-video denoising to directly shape the representation used for action prediction, while exclusive routing prevents the policy from bypassing that representation. At deployment, the world-model branch is removed, leaving only the VLM, World Adapter, and action expert, with no online video-model inference. With a 2B backbone and no embodied action pretraining, World Tokens is highly competitive on LIBERO, attains the best reported averages on SIMPLER, substantially improves real-world R1 Pro success over a matched action-only baseline, and generates each action chunk at VLA-level latency.",
    "submittedDate": "2026-08-10",
    "primaryCategory": "Real-Time / Efficient WAM",
    "secondaryCategories": [],
    "bibtexKey": "TEEPTTM",
    "arxivUrl": "https://arxiv.org/abs/2608.09730",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.09730",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "潜空间预测与JEPA",
      "高效推理与实时控制"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.09516",
    "title": "HarnessWAM: Bridging Prediction and Deliberation in World Action Models",
    "authors": "Gu, Zhaopeng; Zhu, Bingke; Lin, Tianxi; Zhu, Guibo; Chen, Yingying; Wang, Kai; Yuan, Tingyu; Zhao, Chaoyang; Li, Zhaowen; Su, Peng; Wang, Jinqiao",
    "affiliations": "1 Institute of Automation, Chinese Academy of Sciences, Beijing, China; 2 School of Artificial Intelligence, University of Chinese Academy of Sciences, Beijing, China; 3 Yinwang Intelligent Technology Co., Ltd., Shenzhen; 4 Beijing Institute of Technology, Beijing, China; 5 Wuhan AI Research, Wuhan, China",
    "contribution": "To address this gap, we propose HarnessWAM, an agentic framework for WAMs. HarnessWAM achieves state-of-the-art full-task and subtask success rates of 59.6% and 69.9% on RoboMemArena, and an SR of 23.7% on RoboCerebra Ideal.",
    "abstract": "World Action Models (WAMs) jointly learn environmental dynamics and robot actions, introducing priors over physical evolution into embodied control. However, finite-horizon prediction and action generation are insufficient for complex embodied tasks that require global planning, cross-stage state maintenance, execution verification, and failure recovery. We refer to this mismatch as the prediction-deliberation gap of WAMs. To address this gap, we propose HarnessWAM, an agentic framework for WAMs. HarnessWAM employs a vision-language-model-based Task Manager to maintain an evidence-grounded scene belief and a structured task graph. A capability-conditioned executable-space projection further constrains open-ended semantic plans into sequences of atomic skills that satisfy task dependencies, embodiment-state constraints, and the capability boundary of the underlying WAM. During execution, HarnessWAM operates through an event-driven, dual-timescale feedback loop: a lightweight progress estimator continuously provides high-frequency execution evidence, while the Task Manager deliberates at salient milestones by jointly considering the current observation, task state, and interaction history to determine whether to advance the task, acquire additional observations, revise the plan, or initiate local recovery. This mechanism enables the robot to recover its state after a subtask failure and resume execution without discarding previously acquired scene knowledge. HarnessWAM achieves state-of-the-art full-task and subtask success rates of 59.6% and 69.9% on RoboMemArena, and an SR of 23.7% on RoboCerebra Ideal. These results demonstrate that model-external structured state maintenance and closed-loop agentic decision making can effectively extend the local control capabilities of WAMs into embodied task execution that is plannable, verifiable, and recoverable.",
    "submittedDate": "2026-08-10",
    "primaryCategory": "Real-Time / Efficient WAM",
    "secondaryCategories": [],
    "bibtexKey": "HarnessWAM",
    "arxivUrl": "https://arxiv.org/abs/2608.09516",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.09516",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "记忆与长时序",
      "策略后训练与WM-RL"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.09492",
    "title": "Rethink Before You Execute: Adaptive Execution for World Action Models",
    "authors": "Ye, Feng; Zhao, Yiming; Yu, Yong; Zhou, Hongxu; Pan, Yong; Xue, Yuan; Jia, Peng; Jia, Chuanmin",
    "affiliations": "1Wangxuan Institute of Computer Technology, Peking University,2Simplexity Robotics",
    "contribution": "We propose TempoWAM (Timing Execution by Monitoring Progress Online), a lightweight plug-and-play execution scheme for WAMs.",
    "abstract": "World Action Models (WAMs) jointly predict future actions and the evolution of the environment. At each inference, a WAM generates a chunk of actions and the robot executes a fixed prefix before replanning. We argue that this fixed execution horizon is poorly matched to execution dynamics: the chunk reliability varies across task stages, so when to replan depends on the result of accumulated execution, not on the step counts. We propose TempoWAM (Timing Execution by Monitoring Progress Online), a lightweight plug-and-play execution scheme for WAMs. A Recurrent Progress Monitor first estimates task progress from the current observation, task instruction, remaining actions, and execution history; and an Adaptive Execution Protocol then evaluates whether the chunk is advancing the task to decide if replanning is needed. To bridge the training-deployment gap, the protocol is calibrated by a task-dependent calibration factor with online adaptation. Experiments on LIBERO, RoboTwin, and real-world tasks show that TempoWAM consistently improves the efficiency-success trade-off of WAM execution. On real robots, it reduces WAM inferences by 26.9% on easy tasks while maintaining success, and improves success by 13.3 points on difficult tasks.",
    "submittedDate": "2026-08-10",
    "primaryCategory": "Real-Time / Efficient WAM",
    "secondaryCategories": [],
    "bibtexKey": "TempoWAM",
    "arxivUrl": "https://arxiv.org/abs/2608.09492",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.09492",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "记忆与长时序",
      "高效推理与实时控制"
    ],
    "architecture": "不适用",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.09381",
    "title": "JEPA-WAM: Learning Vision-Language-Action Policies with Joint-Embedding World Modeling",
    "authors": "Lin, Yihan; He, Jiawei; Bao, Shifeng; Zhao, Chen; Li, Yang; Wang, Xiaobo; Wang, Yan; Chi, Cheng; Zhang, Jing",
    "affiliations": "1School of Information, Renmin University of China, Beijing, China; 3Key Laboratory of Data Engineering and Knowledge Engineering, Beijing, China; 4Engineering Research Center of Database and Business Intelligence, Beijing, China; 5Shenzhen University of Advanced Technology, Shenzhen, China; 6Institute for AI Industry Research (AIR), Tsinghua University, Beijing, China",
    "contribution": "We introduce JEPA-WAM, a latent WAM built in a pretrained V-JEPA space, which couples latent transition prediction with continuous action generation through a shared predictor. On LIBERO-Plus, JEPA-WAM achieves 79.2%, the best result without large-scale robot-policy pretraining, while its pretrained π0.5π_{0.5} instantiation reaches 86.3%, achieving the best overall performance.",
    "abstract": "Robust robot control benefits from explicitly modeling state transitions, but video-generation world action models (WAMs) introduce substantial deployment cost. Existing latent WAMs avoid explicit future generation, but often compress predictive representations or separate predictive modeling from the representations used for action generation. We introduce JEPA-WAM, a latent WAM built in a pretrained V-JEPA space, which couples latent transition prediction with continuous action generation through a shared predictor. JEPA-WAM predicts a spatially structured joint current-future target that captures task-shared visual temporal structure between current and future observations, while preserving dense patch-level correspondence. Through the shared predictor, transition supervision directly shapes the backbone, from which dedicated representations are extracted for action prediction. The same design can also be instantiated in pretrained VLA policies while preserving their original perception and action pathways. On LIBERO-Plus, JEPA-WAM achieves 79.2%, the best result without large-scale robot-policy pretraining, while its pretrained π0.5π_{0.5} instantiation reaches 86.3%, achieving the best overall performance. Experiments on RoboTwin 2.0 and real-world bimanual manipulation further demonstrate strong generalization under visual and spatial shifts.",
    "submittedDate": "2026-08-10",
    "primaryCategory": "Latent / Representation WAM",
    "secondaryCategories": [],
    "bibtexKey": "JEPAWAM",
    "arxivUrl": "https://arxiv.org/abs/2608.09381",
    "codeUrls": [
      "https://github.com/SpriteWithoutIce/JEPA_WAM"
    ],
    "projectUrl": "https://spritewithoutice.github.io/JEPA_WAM/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.09381",
    "pdfUrl": "https://arxiv.org/pdf/2608.09381",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{JEPAWAM,\n  title={JEPA-WAM: Learning Vision-Language-Action Policies with Joint-Embedding World Modeling},\n  author={Lin, Yihan and He, Jiawei and Bao, Shifeng and Zhao, Chen and Li, Yang and Wang, Xiaobo and Wang, Yan and Chi, Cheng and Zhang, Jing},\n  journal={arXiv:2608.09381},\n  year={2026}\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "潜空间预测与JEPA",
      "泛化与动作对齐"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.08839",
    "title": "SG-WAM: Text-Grounded and Spatial-aware Semantic Guidance for World-Action Models",
    "authors": "He, Junjie; Li, Junfeng; Zhong, Zhide; Yan, Haodong; Li, Ruixin; Zheng, Yangyang; Zhu, Jiaguan; Zhang, Tianran; Du, Yuqiao; Chen, Wen; Zhou, Shunbo; Li, Haoang",
    "affiliations": "1The Hong Kong University of Science and Technology (Guangzhou), Guangzhou, China",
    "contribution": "To overcome this limitation, we propose SG-WAM, a semantic guidance method for world-action models that leverages a vision-language model (VLM) as a semantic planner to enhance the instruction-grounding capacity of world-action models.",
    "abstract": "World-Action Models (WAMs) have emerged as a promising paradigm for robotic manipulation. However, most existing WAMs generate future videos and actions by relying mainly on visual cues rather than language instructions, since off-the-shelf text encoders embed instructions independently of visual observations. As a result, the videos predicted by these WAMs are often semantically misaligned with their corresponding language instructions, which degrades the accuracy of the predicted actions. To overcome this limitation, we propose SG-WAM, a semantic guidance method for world-action models that leverages a vision-language model (VLM) as a semantic planner to enhance the instruction-grounding capacity of world-action models. Specifically, we train a VLM-based planner to predict text-grounded and spatial-aware semantic foresight. The text-grounded semantic foresight grounds the instruction by identifying the correct target objects, and the spatial-aware semantic foresight provides the scene geometry for precise manipulation. We then inject this foresight into the world-action model as high-level semantic guidance, ensuring that both future-video generation and action prediction faithfully follow the language instruction. Extensive experiments in simulation and the real world demonstrate the superiority of our semantic guidance method, showcasing precise manipulation and strong instruction-following capabilities.",
    "submittedDate": "2026-08-09",
    "primaryCategory": "3D/4D WAM",
    "secondaryCategories": [],
    "bibtexKey": "SGWAM",
    "arxivUrl": "https://arxiv.org/abs/2608.08839",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.08839",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "联合视频动作建模",
      "泛化与动作对齐"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "联合预测",
    "quadrant": "Q3 · Dual-system × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.08558",
    "title": "Vid2WAM: Distilling Video Diffusion Priors into World Action Models",
    "authors": "Qiu, Chenhao; Wang, Ruixiang; Zhao, Runyi; Lin, Sixu; Gu, Songen; Nan, Shufeng; Liu, Guiliang; Jia, Kui; Fu, Yanwei; Wu, Simo",
    "affiliations": "1Fudan University; 2The Chinese University of Hong Kong, Shenzhen; 3Shanghai Innovation Institute",
    "contribution": "In this paper, we propose Vid2WAM, an offline distillation framework that transfers visual diffusion priors from a large video foundation model into a compact WAM student. To robustly integrate synthetic and real supervision, we introduce source-aware residual action adaptation that learns source-specific corrections around a shared action backbone and mitigates interference from noisy pseudo-actions.",
    "abstract": "World Action Models (WAMs) improve robot policy learning by jointly modeling future visual dynamics and actions. However, their scalability and generalization remain constrained by their reliance on costly expert demonstrations. We challenge this by asking whether future supervision for WAMs must originate from target-task expert trajectories. In this paper, we propose Vid2WAM, an offline distillation framework that transfers visual diffusion priors from a large video foundation model into a compact WAM student. Given an observation and language instruction, Vid2WAM distills supervision through two complementary channels: task-conditioned future rollouts directly supervise the student's future prediction branch, while an inverse dynamics model recovers embodiment-specific pseudo-actions for action learning. To robustly integrate synthetic and real supervision, we introduce source-aware residual action adaptation that learns source-specific corrections around a shared action backbone and mitigates interference from noisy pseudo-actions. During inference, both the video teacher and inverse dynamics model are discarded, leaving only the WAM student for efficient deployment. Simulation and real-world experiments demonstrate that Vid2WAM improves novel-task generalization and data efficiency under limited expert demonstrations while preserving low-latency inference.",
    "submittedDate": "2026-08-09",
    "primaryCategory": "Real-Time / Efficient WAM",
    "secondaryCategories": [],
    "bibtexKey": "Vid2WAM",
    "arxivUrl": "https://arxiv.org/abs/2608.08558",
    "codeUrls": [
      "https://github.com/qch-FA/Vid2WAM"
    ],
    "projectUrl": "https://qch-fa.github.io/vid2wam-website/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.08558",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "泛化与动作对齐",
      "高效推理与实时控制"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.06375",
    "title": "ω-0: A Latent Predictive World Action Model for Concurrent Humanoid Loco-Manipulation",
    "authors": "Li, Zhe; Zhang, Zhenzhe; Wei, Yangyang; Zhang, Wenjie; Yuan, Xichen; Zhi, Peiyuan; Li, Gen; Guo, Xinying; Gao, Fengjie; Yang, Jianfei; Zhang, Shanghang",
    "affiliations": "1MARS Lab, NTU,2PKU, 3BAAI, 4HKUST(GZ)",
    "contribution": "We present ωω-0, a latent predictive whole-body world-action model for real-world humanoid concurrent loco-manipulation. Real-world experiments on 11 household tasks demonstrate that a single ωω-0 model can produce smooth manipulate-while-moving behaviors and consistently outperform representative imitation learning, VLA, humanoid, and WAM baselines.",
    "abstract": "Humanoid household tasks often require concurrent loco-manipulation, where the robot must move, adjust posture, maintain balance, and manipulate objects as a single coordinated behavior. Yet existing humanoid policies typically decompose locomotion and manipulation, while recent world-action models remain either arm-centric or video-centered. We present ωω-0, a latent predictive whole-body world-action model for real-world humanoid concurrent loco-manipulation. Given a language instruction, current visual observation, and robot proprioceptive state, ωω-0 directly predicts controller-compatible whole-body action latents for real-robot execution. Rather than reconstructing future videos, ωω-0 learns compact future observation embeddings as a lightweight predictive objective, coupling latent visual foresight with diffusion-based whole-body action generation. The model supports egocentric RGB, exocentric RGB, and exocentric depth inputs, and leverages controller-based simulation replay to ground human/public visual-motion priors into robot-executable action latents. We further collect ωω-HOME, a 40+ hour real-world household humanoid dataset with synchronized multi-view observations, whole-body SMPL motions, robot states, and action latents. Real-world experiments on 11 household tasks demonstrate that a single ωω-0 model can produce smooth manipulate-while-moving behaviors and consistently outperform representative imitation learning, VLA, humanoid, and WAM baselines.",
    "submittedDate": "2026-08-09",
    "primaryCategory": "Latent / Representation WAM",
    "secondaryCategories": [
      "3D/4D WAM",
      "Multimodal / Tactile WAM",
      "Navigation / Driving / Domain WAM",
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "Omega0",
    "arxivUrl": "https://arxiv.org/abs/2608.06375",
    "codeUrls": [],
    "projectUrl": "https://gentlefress.github.io/OMEGA-0_page/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.06375",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "潜空间预测与JEPA",
      "视觉规划与IDM",
      "导航"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2607.02501",
    "title": "Embodied.cpp: A Portable Inference Runtime of Embodied AI Models on Heterogeneous Robots",
    "authors": "Xu, Ling; Li, Borui; Wu, Hao; Han, Chuyu; Li, Xiangyu; Hua, Mohan; Jiang, Shiqi; Cao, Ting; Li, Chuanyou; Zhong, Sheng; Wang, Shuai",
    "affiliations": "Southeast University 2; Nanjing University 3; Microsoft Research; Institute for AI Industry Research (AIR), Tsinghua University",
    "contribution": "We present Embodied..cpp, a portable C++ inference runtime for embodied models. Overall, Embodied..cpp achieves 1.05x-2.70x inference speedups and 7\\%-77\\% lower VRAM relative to Python baselines, while maintaining near-baseline success for most configurations.",
    "abstract": "Embodied AI models now span vision-language-action (VLA) models and world-action models (WAMs), but practical deployment remains fragmented across model-specific Python stacks, backend assumptions, and robot-side glue code, especially on heterogeneous edge devices. Existing inference runtimes are designed mainly for request-response serving and therefore do not satisfy the runtime contract of embodied deployment: multi-rate execution inside closed-loop control, latency-first batch-1 inference on heterogeneous hardware, and extensible embodied interfaces beyond fixed token I/O. We present Embodied..cpp, a portable C++ inference runtime for embodied models. Based on an architectural analysis of representative VLA models and WAMs, Embodied..cpp captures a shared execution path and organizes it into five layers: input adapters, sequence builders, backbone execution, head plugins, and deployment adapters. The runtime provides modular multi-rate execution, latency-first fused inference, and extensible operator and I/O support, enabling deployment across heterogeneous devices, robots, and simulators through one backend abstraction. We evaluate Embodied..cpp on three VLA and two WAM models, using normalized comparisons across Python and C++ quantization configurations. Overall, Embodied..cpp achieves 1.05x-2.70x inference speedups and 7\\%-77\\% lower VRAM relative to Python baselines, while maintaining near-baseline success for most configurations. These results show that Embodied..cpp improves deployment efficiency while preserving high control quality across diverse embodied model architectures. Project Link: https://github.com/SEU-PAISys/Embodied.cpp",
    "submittedDate": "2026-08-09",
    "primaryCategory": "Real-Time / Efficient WAM",
    "secondaryCategories": [
      "Evaluation / Survey / Theory"
    ],
    "bibtexKey": "Embodiedcpp",
    "arxivUrl": "https://arxiv.org/abs/2607.02501",
    "codeUrls": [
      "https://github.com/SEU-PAISys/Embodied.cpp"
    ],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2607.02501",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "Related resources",
    "subcategories": [
      "综述与技术资源",
      "训练优化与蒸馏"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2607.24744",
    "title": "Data Pyramid for Embodied Manipulation: A Survey",
    "authors": "Yifan Ye; Yankai Fu; Yaoxu Lv; Bohan Hou; Jun Cen; Lingdong Kong; Duo Zheng; Tianxing Chen; Jiaming Liu; Ziang Cao; Yunfan Lou; Wei Chow; Xian Sun; Yingshuo Wang; Kuangzhi Ge; Xiaowei Chi; Xidong Zhang; Zhibo Pang; Yiwu Zhong; Sirui Han; Zhihe Lu; Weihao Yuan; Qifeng Chen; Michael Yu Wang; Yao Mu; Ziwei Liu; Jianfei Yang; Ping Luo; Shanghang Zhang",
    "affiliations": "Not identified",
    "contribution": "Multimodal foundation models learned to see and to speak by consuming the whole internet. Embodied agents admit no such shortcut, since they require data that couple observations with physical states and actions.",
    "abstract": "Multimodal foundation models learned to see and to speak by consuming the whole internet. Embodied agents admit no such shortcut, since they require data that couple observations with physical states and actions. These signals can be provided, to varying degrees, by multiple data sources. In this work, we organize the embodied data ecosystem as a \"pyramid\" spanning five complementary sources: real-robot data, UMI-style data, egocentric and exocentric data, simulation data, and general vision-language data. We organize the pyramid around the tension between scalability and robot alignment, and further characterize each source in terms of data quality, diversity, reusability, and physical fidelity. We then analyze recent embodied foundation models through the lens of their data recipes, examining how different sources are selected, aligned, and mixed during pretraining. For embodied brain models, vision-language-action models, and world-action models alike, we relate data composition to capabilities in perception, reasoning, planning, action generation, and world prediction. We close by discussing six open challenges: building large-scale tactile datasets, collecting failure and recovery data, developing scalable data-collection pipelines, aligning actions across embodiments, leveraging egocentric data for dexterous manipulation, and designing principled data recipes for robot learning. We hope this work paves the foundation for the design of next-generation embodied systems.",
    "submittedDate": "2026-08-08",
    "primaryCategory": "Evaluation / Survey / Theory",
    "secondaryCategories": [
      "Multimodal / Tactile WAM"
    ],
    "bibtexKey": "DataPyramid",
    "arxivUrl": "https://arxiv.org/abs/2607.24744",
    "codeUrls": [
      "https://github.com/worldbench/awesome-embodied-data-pyramid"
    ],
    "projectUrl": "https://jasper-aaa.github.io/embodied-data-pyramid/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2607.24744",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "Related resources",
    "subcategories": [
      "综述与技术资源",
      "数据集与数据采集"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.21402",
    "title": "Selective Cross-View Consistency for World Action Models: Held-Out Viewpoint Robustness Without Test-Time Camera Information",
    "authors": "Huang, Bingqi; Wei, Bingchuan; Cai, Yingkai; Wang, Zhaokui",
    "affiliations": "Tsinghua University",
    "contribution": "We introduce a carve-and-hold-out evaluation protocol on the LIBERO-Plus camera track that separates a distribution-matched ceiling from genuine interpolation and extrapolation to held-out viewpoints, with a matched pair-trained control isolating the effect of the consistency term from pair exposure.",
    "abstract": "World action models (WAMs) jointly denoise future video frames and robot actions, and the video prior is expected to generalize their control. Camera viewpoint change remains one of their hardest perturbation axes. We study a question specific to this model class: when training with same-state cross-view image pairs, on which output coordinates should a consistency loss be imposed? The WAM denoising target mixes view-covariant coordinates, namely the predicted future scene, with view-invariant coordinates, namely the action chunk, future proprioception, and value. We show that consistency applied to the covariant block is provably harmful, shrinking legitimate view-specific content to a fraction 1/(1+4λ)1/(1+4λ) of its true value, and we verify this shrinkage law in controlled experiments. Selective cross-view consistency (SCVC) therefore constrains only the invariant block, requires no camera labels, extrinsics, depth, or view synthesis at training or test time, and leaves the deployment interface unchanged. We introduce a carve-and-hold-out evaluation protocol on the LIBERO-Plus camera track that separates a distribution-matched ceiling from genuine interpolation and extrapolation to held-out viewpoints, with a matched pair-trained control isolating the effect of the consistency term from pair exposure. On held-out orbital viewpoints beyond the training envelope, SCVC improves closed-loop success over the matched control by 12.2 points (95% CI [7.4, 17.0]; +15.5, CI [11.7, 19.4], under an independent second seed) -- an effect two further camera axes replicate -- while interpolation within the envelope shows no gain in either seed (-1.2 and -4.3 points) and in-distribution competence is preserved (-0.6, -0.2). We also report a cross-backbone audit showing that published camera-robustness numbers are confounded by wrist-camera pose stability.",
    "submittedDate": "2026-08-07",
    "primaryCategory": "3D/4D WAM",
    "secondaryCategories": [
      "Evaluation / Survey / Theory",
      "Multimodal / Tactile WAM"
    ],
    "bibtexKey": "SCVC",
    "arxivUrl": "https://arxiv.org/abs/2608.21402",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.21402",
    "pdfUrl": "https://arxiv.org/pdf/2608.21402",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{SCVC,\n  title={Selective Cross-View Consistency for World Action Models: Held-Out Viewpoint Robustness Without Test-Time Camera Information},\n  author={Huang, Bingqi and Wei, Bingchuan and Cai, Yingkai and Wang, Zhaokui},\n  journal={arXiv:2608.21402},\n  year={2026}\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "联合视频动作建模",
      "三维多视角建模",
      "泛化与动作对齐"
    ],
    "architecture": "One Model",
    "predictionParadigm": "联合预测",
    "quadrant": "Q1 · One Model × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.06994",
    "title": "Decoupling Intention from Trajectory: A Representational Deduction Framework for World Action Models",
    "authors": "Ma, Xiangkai; Ma, Yue; Wang, Junjie; Xu, Sheng; Li, Mingyang; Zhang, Han; Zhuang, Yuzheng; Li, Wenzhong; Yuan, Zhihao",
    "affiliations": "Joy Future Academy, JD",
    "contribution": "We propose PILOT (Physical Inference for Latent Optimized Trajectories), whose core Representational Deduction (RD) bridges this gap by integrating motion thought-of-chain (CoT) guidance as a native model capability.",
    "abstract": "World Action Models (WAMs) aim to construct a unified architecture capable of understanding world state evolution and guiding to generative motion planning. However, existing visual branches focus on predicting static visual observation, rather than reflecting potential transition information that captures the evolution of world states under motion interactions. This leads to representational entanglement between high-level physical condition evolution and low-level action trajectory generation within the Action Model, creating a structural bottleneck while weakening the predictive capability of world evolution modeling for action generation. We propose PILOT (Physical Inference for Latent Optimized Trajectories), whose core Representational Deduction (RD) bridges this gap by integrating motion thought-of-chain (CoT) guidance as a native model capability. Specifically, RD aims to encourage the action branch to explicitly model potential state transition tokens, which are retained as CoT in the reasoning space to guide fine-grained motion trajectory. Experiments demonstrate that RD not only significantly improves the success rate and generalization ability of WAMs in complex robotic manipulation tasks but also enhances the model's physical interpretability by decoupling high-level motion semantics from low-level trajectory details. Furthermore, the abundant state transition supervision signals introduced by RD effectively alleviate the sparse supervision in action generation, enabling it to serve as an efficient few-shot real-robot fine-tuning strategy and demonstrating superior scalability for migration to mainstream WAM architectures.",
    "submittedDate": "2026-08-07",
    "primaryCategory": "Latent / Representation WAM",
    "secondaryCategories": [
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "RD",
    "arxivUrl": "https://arxiv.org/abs/2608.06994",
    "codeUrls": [],
    "projectUrl": "https://pilot-wam-2026.github.io/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.06994",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "潜空间预测与JEPA",
      "高效推理与实时控制"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.05903",
    "title": "Robust-WAM: Bridging Generative Pretraining and Semantic Foresight in World-Action Models",
    "authors": "Yan, Haodong; Li, Junfeng; He, Junjie; Zhong, Zhide; Yu, MingMing; Song, Wenxuan; Zhu, Jiaguan; Zheng, Yangyang; Du, Yuqiao; You, Jiadi; Cai, Yingjie; Yan, Xu; Zhao, Guanyi; Liu, Bingbing; Li, Haoang",
    "affiliations": "1The Hong Kong University of Science and Technology (Guangzhou), Guangzhou, China,2Beihang; University, Beijing, China,3Huawei Foundation Model Department; over time (Team Wan et al., 2025; NVIDIA et al., 2025)",
    "contribution": "To overcome this dilemma, we propose Robust-WAM, a general post-training method for video-generation-based WAMs that preserves the VAE-based generative path and adds a lightweight semantic foresight alignment objective on the action stream.",
    "abstract": "Mainstream World-Action Models (WAMs) adapt pretrained video generation models (VGMs) for robot control, transferring their learned dynamics prior for action prediction. These VGMs are typically trained in a variational autoencoder (VAE) latent space. However, the VAE latent space is optimized for pixel reconstruction, which rewards fine appearance detail and leaves the action prediction fragile under visual shifts. Recent works build WAMs in semantic latent space, which are more robust to appearance shifts. However, these models cannot leverage the large-scale VGM pretraining that exists only in VAE space. To overcome this dilemma, we propose Robust-WAM, a general post-training method for video-generation-based WAMs that preserves the VAE-based generative path and adds a lightweight semantic foresight alignment objective on the action stream. This retains the large-scale VGM pretraining while grounding actions in appearance-invariant dynamics that stay reliable under illumination shifts and other visual out-of-distribution conditions. Specifically, we employ learnable query tokens to bring future-scene semantics into the action stream by aligning their output hidden states with the semantic foresight of future ground-truth frames. To establish the temporal correspondence between each query and the future step it describes, we give it the positional encoding of the matching action tokens. Experiments on out-of-distribution generalization simulation benchmarks and a real-robot setup show that our Robust-WAM consistently improves the success rates of multiple WAM baselines without sacrificing in-distribution performance.",
    "submittedDate": "2026-08-07",
    "primaryCategory": "Latent / Representation WAM",
    "secondaryCategories": [
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "RobustWAM",
    "arxivUrl": "https://arxiv.org/abs/2608.05903",
    "codeUrls": [
      "https://github.com/Haodong-Yan/Robust-WAM-release"
    ],
    "projectUrl": "https://haodong-yan.github.io/robust-wam-project-page/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.05903",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "潜空间预测与JEPA",
      "泛化与动作对齐"
    ],
    "architecture": "不适用",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.06008",
    "title": "Adaptive-WAM: Quality-Guided Early-Exit Planning from Intermediate Video-Diffusion Features",
    "authors": "Ang, Sining; Yang, Yuguang; Wang, Yan",
    "affiliations": "1Institute for AI Industry Research (AIR), Tsinghua University; 2Department of Automation, University of Science and Technology of China; 3School of Electronic Information Engineering, Beihang University",
    "contribution": "Based on this observation, we introduce Adaptive-WAM, a quality-aware multi-exit planner built on a Wan2.2-5B backbone. On NAVSIM, the adaptive single-trajectory planner achieves 90.8 PDMS; a separate fixed-exit variant reaches 92.6 PDMS with 64 proposals.",
    "abstract": "Large video diffusion models provide rich spatiotemporal priors for autonomous driving, but existing world-action models often inherit the cost of iterative future-video generation even though deployment only requires an ego trajectory. We ask a more basic question: how much of a video diffusion model must be executed to make a reliable driving decision? Through a controlled study of video denoising timesteps and Diffusion Transformer (DiT) depth, we find that planning performance is largely insensitive to the tested video-noise levels, whereas strong trajectories can already be decoded from intermediate layers. Based on this observation, we introduce Adaptive-WAM, a quality-aware multi-exit planner built on a Wan2.2-5B backbone. Trajectory diffusion heads are attached to selected DiT blocks, and a lightweight trajectory-quality scorer terminates inference once the best trajectory decoded so far satisfies a quality threshold; otherwise, computation continues from the cached hidden state to a deeper exit. The deployed planner therefore avoids the iterative classifier-free denoising loop and VAE decoding required for future-video synthesis, while dynamically allocating backbone depth according to trajectory quality. On NAVSIM, the adaptive single-trajectory planner achieves 90.8 PDMS; a separate fixed-exit variant reaches 92.6 PDMS with 64 proposals. It further obtains 89.9 EPDMS on NAVSIM v2, yielding the best reported results among the compared front-view video world-model planners. Without target-domain fine-tuning, Adaptive-WAM transfers to nuScenes with 0.88 m average L2 error and a 0.08\\% collision rate. On an A100, adaptive routing improves PDMS from 90.62 to 90.79 while averaging 170 ms end-to-end planning latency, approximately 10\\% below the 190 ms fixed block-15 planner and 47\\% below the 320 ms fixed full-depth planner. Code will be released.",
    "submittedDate": "2026-08-06",
    "primaryCategory": "3D/4D WAM",
    "secondaryCategories": [
      "Evaluation / Survey / Theory",
      "Navigation / Driving / Domain WAM",
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "AdaptiveWAM",
    "arxivUrl": "https://arxiv.org/abs/2608.06008",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.06008",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "自动驾驶",
      "高效推理与实时控制"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "联合预测",
    "quadrant": "Q3 · Dual-system × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.04657",
    "title": "MobileWAM: Bridging World Action Models to Mobile Manipulation with Chain-of-Foresight",
    "authors": "Fan, Zehua; He, Junjie; Song, Wenxuan; Wang, Xi; Lyu, Wenqi; Zhao, Linge; Li, Fuhao; You, Zihan; Yang, Yifei; Xu, Kaiming; Jiang, Qi; Jiang, Yue; Li, Haoang; Chi, Cheng; Gao, Feng; Li, Bailin; Wang, Yan",
    "affiliations": "1Institute for AI Industry Research (AIR), Tsinghua University,2Shanghai Jiao Tong University; 3The Hong Kong University of Science and Technology (Guangzhou),4AIR Wuxi Innovation Center; Tsinghua University,5The University of Adelaide,6Wuhan University,7Southeast University; 8Beijing Jiaotong University,9Fudan University,10Li Auto, 11School of Information, Renmin; University of China",
    "contribution": "MobileWAM surpasses state-of-the-art mobile manipulation policies on ManiSkill-HAB and fine-tunes to a real ARX Lift2 mobile manipulator across diverse tasks with strong generalization.",
    "abstract": "World action models (WAMs) built on video generation backbones are a rising recipe for robot learning, yet remain confined to tabletop manipulation. Mobile manipulation demands simultaneous locomotion and whole-body manipulation amid scene-scale dynamics, yet is still dominated by dynamics-blind visual encoders with hand-crafted coordination. We bridge this gap with MobileWAM, a mixture-of-transformers architecture that fuses a pretrained video diffusion transformer with a lightweight action expert through layerwise joint attention, translating internet-scale motion priors into whole-body control. To reconcile the heterogeneous dynamics of moving and manipulating, each feed-forward layer of the action expert becomes a three-expert mixture of shared, locomotion, and manipulation experts, softly routed by the motion intent in the action tokens. To densify supervision, we further propose Chain-of-Foresight (CoF): intermediate representations sequentially predict a chain of future latent chunks, each step conditioned on its predecessor. CoF pairs naturally with our decoupled video--action denoising scheme. At deployment, the WAM serves as a pure current-frame encoder; foresight acts only through gradients, so at inference the foresight chain and video generation are discarded, leaving only policy-level cost. MobileWAM surpasses state-of-the-art mobile manipulation policies on ManiSkill-HAB and fine-tunes to a real ARX Lift2 mobile manipulator across diverse tasks with strong generalization. Code will be released soon.",
    "submittedDate": "2026-08-06",
    "primaryCategory": "Navigation / Driving / Domain WAM",
    "secondaryCategories": [
      "Latent / Representation WAM",
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "MobileWAM",
    "arxivUrl": "https://arxiv.org/abs/2608.04657",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.04657",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "导航",
      "记忆与长时序",
      "高效推理与实时控制"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.00793",
    "title": "DynamicWAM: Dual-Path Motion Conditioning for World-Action Models in Dynamic Manipulation",
    "authors": "Lou, Yunfan; Gao, Hewen; Zhu, Xiyu; Qiao, Zhuoran; Han, Xuan; Yang, Yifan; Ye, Yifan; Yao, Boxian; Pang, Zhibo",
    "affiliations": "1PKU-PI Lab 2Peking University 3ePyBot Intelligence 4National University of Singapore 5Tsinghua University",
    "contribution": "We propose DynamicWAM, a compact WAM for dynamic object manipulation with dual-path motion conditioning. On DOMINO, DynamicWAM achieves a 38.2% success rate and a 53.2 manipulation score, outperforming all evaluated baselines.",
    "abstract": "Dynamic manipulation requires robots to infer target motion and respond promptly, yet existing World-Action Models (WAMs) typically condition only on the current frame and execute large backbones synchronously, limiting motion awareness and responsive control in dynamic scenes. We propose DynamicWAM, a compact WAM for dynamic object manipulation with dual-path motion conditioning. DynamicWAM introduces history-flow conditioning, encoding temporally aligned optical-flow frames alongside the current observation through a frozen pretrained video VAE to preserve spatial motion structure, while injecting kinematic descriptors of displacement, duration, velocity, and acceleration into the action expert to provide motion magnitude and timing. The two complementary paths are fused through joint world-action attention. A distilled compact backbone and real-time chunking (RTC)-based asynchronous execution further enable responsive control. On DOMINO, DynamicWAM achieves a 38.2% success rate and a 53.2 manipulation score, outperforming all evaluated baselines. Across 12 real-world tasks spanning linear, circular, and compound target motion, it achieves a 46.7% average success rate, exceeding the strongest baseline by 22.9 percentage points.",
    "submittedDate": "2026-08-06",
    "primaryCategory": "Real-Time / Efficient WAM",
    "secondaryCategories": [],
    "bibtexKey": "DynamicWAM",
    "arxivUrl": "https://arxiv.org/abs/2608.00793",
    "codeUrls": [
      "https://github.com/Autumn1337/DynamicWAM"
    ],
    "projectUrl": "https://dynamicwam.github.io/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.00793",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "高效推理与实时控制",
      "联合视频动作建模",
      "泛化与动作对齐"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "联合预测",
    "quadrant": "Q3 · Dual-system × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.04996",
    "title": "DreamWAM: Beyond RGB Future Prediction for World Action Models",
    "authors": "Yuan, Shanglin; Zhao, Weiheng; Shi, Xin; Jiang, Haoyi; Guo, Xianda; Liu, Liu; Liu, Wenyu; Sui, Wei; Wang, Xinggang",
    "affiliations": "1Huazhong University of Science and Technology2D-Robotics 3Wuhan University 4Horizon Robotics",
    "contribution": "We introduce DreamWAM, which reformulates future prediction as structured world modeling beyond RGB, representing future states through complementary views of appearance, motion, geometry, and semantics.",
    "abstract": "World Action Models (WAMs) learn action-relevant representations by predicting how the observed world will evolve. Most existing WAMs define this future in RGB space, where task-relevant state transitions are entangled with nuisance variations in texture, illumination, background, and viewpoint. We argue that WAMs should explicitly predict action-relevant future state rather than relying on RGB prediction alone. We introduce DreamWAM, which reformulates future prediction as structured world modeling beyond RGB, representing future states through complementary views of appearance, motion, geometry, and semantics. During training, DreamWAM combines joint latent denoising of RGB and motion with lightweight gated residual branches for geometry and semantics. Shared attention between VideoDiT and ActionDiT allows the action branch to learn from these future-state predictions, while all beyond-RGB supervision branches are disabled at inference and deployment remains RGB-only. Across both no-rollout and joint video-action inference, DreamWAM consistently improves the matched RGB-only baselines on LIBERO, from 97.30\\% to 98.40\\% and from 98.00\\% to 98.90\\%, respectively. The gains become larger under unseen LIBERO-Plus perturbations, from 51.36\\% to 63.44\\% and from 69.16\\% to 75.47\\%. The same robustness extends to real-world manipulation, where DreamWAM attains an average success rate of 74.4\\% across unseen changes in lighting, background, and object layout, compared with 55.6\\% for Fast-WAM-Joint. These results show that robust world-action learning depends not only on predicting the future, but on representing it in a form that matters for action. The code and models are publicly released at https://github.com/hustvl/DreamWAM.",
    "submittedDate": "2026-08-05",
    "primaryCategory": "3D/4D WAM",
    "secondaryCategories": [
      "Latent / Representation WAM",
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "DreamWAM",
    "arxivUrl": "https://arxiv.org/abs/2608.04996",
    "codeUrls": [
      "https://github.com/hustvl/DreamWAM"
    ],
    "projectUrl": "https://hustvl.github.io/DreamWAM/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.04996",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "联合视频动作建模",
      "三维多视角建模",
      "泛化与动作对齐"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "联合预测",
    "quadrant": "Q3 · Dual-system × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.04404",
    "title": "Faster-WAM: Efficient Inference-Time Future Conditioning for Robust World Action Models",
    "authors": "Zhao, Weiheng; Jiang, Haoyi; Shi, Xin; Liu, Liu; Huang, Fan; Su, Zhizhong; Sui, Wei; Wang, Xinggang",
    "affiliations": "1Huazhong University of Science and Technology; 2D-Robotics, 3Horizon Robotics,4Xiamen University",
    "contribution": "Specifically, we propose SparseMoT to replace ubiquitous layer-wise fusion with selective video-action interaction at a compact subset of network stages, and Interval KV-Fusion to aggregate multi-depth future representations without increasing attention complexity. Experiments demonstrate that Faster-WAM achieves a substantially better performance-efficiency trade-off than existing WAMs.",
    "abstract": "World Action Models (WAMs) improve robot manipulation by learning how the environment evolves beyond the current observation. However, existing approaches face a fundamental dilemma: Joint-WAMs preserve future-aware representations during inference but incur prohibitive computation costs, while efficient alternatives remove future modeling at inference time and may lose the robustness benefits of temporal reasoning. In this work, we revisit the role of future representations in WAMs and show that inference-time future conditioning is critical for generalization under distribution shifts. This observation motivates Faster-WAM, an efficient future-conditioning WAM that preserves future representations while avoiding expensive video-action interaction. Faster-WAM introduces a sparse future-conditioning framework that computes future representations once and selectively reuses them throughout action denoising. Specifically, we propose SparseMoT to replace ubiquitous layer-wise fusion with selective video-action interaction at a compact subset of network stages, and Interval KV-Fusion to aggregate multi-depth future representations without increasing attention complexity. Experiments demonstrate that Faster-WAM achieves a substantially better performance-efficiency trade-off than existing WAMs. On the out-of-distribution LIBERO-Plus benchmark, Faster-WAM improves success rate from 49.14% to 73.57% compared with Fast-WAM, while running 2.21×\\times faster than Joint-WAM. It further achieves state-of-the-art performance on LIBERO and RoboTwin 2.0, while demonstrating strong robustness in real-world manipulation.",
    "submittedDate": "2026-08-04",
    "primaryCategory": "Real-Time / Efficient WAM",
    "secondaryCategories": [
      "3D/4D WAM",
      "Evaluation / Survey / Theory"
    ],
    "bibtexKey": "FasterWAM2",
    "arxivUrl": "https://arxiv.org/abs/2608.04404",
    "codeUrls": [
      "https://github.com/hustvl/FasterWAM"
    ],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.04404",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "视觉规划与IDM",
      "高效推理与实时控制"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.03701",
    "title": "LiLa-WAM: Lightweight Latent Reasoning World-Action Model for Robotic Manipulation",
    "authors": "Yang, Fan; Su, Yuting; Wang, Xiaobo; You, Yuncheng; Fan, Fugui; Wu, Yuting; Wu, Minghui; Zhao, Chenxu; Ning, JiaHong; Jing, Peiguang",
    "affiliations": "Tianjin University; Shenzhen University of Advanced Technology; Mininglamp Technology; Ministry of Natural Resources Information Center; Sangfor Technologies Inc",
    "contribution": "In this work, we propose LiLa-WAM, a lightweight world-action model that reasons about the future in a compact latent space and can be trained end-to-end on a single 24GB GPU.",
    "abstract": "World-action modeling has emerged as a promising paradigm for robotic control, as it empowers models to go beyond reacting to observations and anticipate how a scene will evolve. However, existing WAMs often incur substantial computational overhead. Pixel-space methods often allocate substantial capacity to visual details that may not be directly relevant to control, while some latent-space methods require multi-stage training to construct the reasoning space. The resulting training cost can make such methods difficult to train under modest computational budgets. In this work, we propose LiLa-WAM, a lightweight world-action model that reasons about the future in a compact latent space and can be trained end-to-end on a single 24GB GPU. Its core design is a compact latent reasoning space jointly shaped by future-state prediction and action generation, which keeps the model lightweight while remaining well aligned with control. For task specification, we further propose the Visual Transition Token(VTT), a language-free task representation that encodes each task as a direction in visual feature space. Experiments on RoboTwin~2.0, LIBERO, and real-robot tasks demonstrate LiLa-WAM's effectiveness, achieving 90.48\\% success across 50 RoboTwin tasks with single-GPU training.",
    "submittedDate": "2026-08-04",
    "primaryCategory": "Latent / Representation WAM",
    "secondaryCategories": [
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "LiLaWAM",
    "arxivUrl": "https://arxiv.org/abs/2608.03701",
    "codeUrls": [
      "https://github.com/teee000/LiLa-WAM"
    ],
    "projectUrl": "https://teee000.github.io/LiLa-WAM-page/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.03701",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "潜空间预测与JEPA",
      "联合视频动作建模",
      "高效推理与实时控制"
    ],
    "architecture": "One Model",
    "predictionParadigm": "联合预测",
    "quadrant": "Q1 · One Model × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.03244",
    "title": "UniNav: A Unified World-Action Diffusion Model for Visual Navigation",
    "authors": "Zhou, Changqing; Luo, Yueru; Jiang, Zeyu; Chen, Changhao",
    "affiliations": "1The Hong Kong University of Science and Technology (Guangzhou); 2The Chinese University of Hong Kong, Shenzhen",
    "contribution": "We present UniNav, a unified world-action model that generates future visual observations and continuous waypoint trajectories through a single diffusion process. Based on this unified framework, we introduce two variants: UniNav-Full jointly predicts interpretable future observations and their corresponding trajectories, while UniNav-Fast removes future-image tokens at inference for efficient trajectory prediction.",
    "abstract": "Image-goal visual navigation is a fundamental capability for embodied agents. Existing navigation policies efficiently predict waypoint trajectories but lack visual foresight, while navigation world models can anticipate future observations but often require costly planning rollouts. We present UniNav, a unified world-action model that generates future visual observations and continuous waypoint trajectories through a single diffusion process. Given history frames and a goal image, UniNav jointly denoises visual and waypoint tokens within a single transformer, unifying future prediction and action generation in a shared framework. To improve spatial grounding, we incorporate geometry-aware camera tokens. We also train on both trajectory-labeled navigation data and video-only data, enabling the model to benefit from diverse videos without waypoint annotations. Based on this unified framework, we introduce two variants: UniNav-Full jointly predicts interpretable future observations and their corresponding trajectories, while UniNav-Fast removes future-image tokens at inference for efficient trajectory prediction. Experiments on navigation benchmarks show that UniNav outperforms the strongest baseline in ATE across all datasets. With one-step inference, UniNav-Fast achieves a latency of 0.1s without a substantial accuracy drop. Code will be released.",
    "submittedDate": "2026-08-04",
    "primaryCategory": "Navigation / Driving / Domain WAM",
    "secondaryCategories": [
      "3D/4D WAM",
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "UniNav",
    "arxivUrl": "https://arxiv.org/abs/2608.03244",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.03244",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "导航",
      "联合视频动作建模",
      "三维多视角建模"
    ],
    "architecture": "One Model",
    "predictionParadigm": "联合预测",
    "quadrant": "Q1 · One Model × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.02578",
    "title": "CoWAM: Coordination Contracts for Selective Policy Intervention with WAMs",
    "authors": "Liu, Shuaijun; Wen, Qifu; Hao, Shuyang; Luo, Qi; Zhang, Chenglong; You, Feiyang; Wu, Chengyu; Su, Ningxin",
    "affiliations": "1The Hong Kong University of Science and Technology (Guangzhou); 2Boston University 3Shanghai Jiao Tong University",
    "contribution": "We present CoWAM, a selective intervention layer that expresses synchronization, role compatibility, and collision convergence as coordination contracts.",
    "abstract": "World Action Models (WAMs) augment robot policies with action-conditioned predicted futures, but a plausible future alone does not justify changing the action that a bimanual policy would execute. We present CoWAM, a selective intervention layer that expresses synchronization, role compatibility, and collision convergence as coordination contracts. Each contract combines typed admissibility checks with event-conditioned verification and calibrated intervention gates. CoWAM preserves the nominal action unless an alternative satisfies every active obligation and provides a clear, low-risk improvement; when the nominal action is also inadmissible, it invokes a predefined abstention fallback. To separate selector quality from proposal quality, all methods operate on identical candidate pools and commit their decisions before shared oracle labeling. Across eight simulated bimanual tasks, CoWAM improves coordination-valid selection by 16.7 percentage points over the contract-only variant and raises closed-loop success by 9.6 percentage points over the strongest selective baseline, while keeping harmful interventions below 1%. Together, these results establish coordination contracts as an effective interface for conservative policy intervention with predicted world-action evidence across coordination-rich bimanual tasks.",
    "submittedDate": "2026-08-03",
    "primaryCategory": "General WAM",
    "secondaryCategories": [],
    "bibtexKey": "CoWAM",
    "arxivUrl": "https://arxiv.org/abs/2608.02578",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.02578",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "策略后训练与WM-RL",
      "泛化与动作对齐"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.02365",
    "title": "Faster-WAM: Do World Action Models Need Deep Action Modules?",
    "authors": "Ma, Liheng; Yang, Rui Heng; Zhang, Zhanguang; Clemente, Mateo; Hu, Ziwen; Cao, Tongtong; Zhang, Yingxue",
    "affiliations": "1 Huawei Noah's Ark Lab; 3 Department of Foundation Model, 2012 Labs",
    "contribution": "To address this limitation, we introduce Dock of Transformer (DoT), a video-centric design principle that treats a pretrained video Transformer as a representation hub and connects lightweight output-heads through docking interfaces. Without additional embodied pretraining, Faster-WAM achieves competitive performance on LIBERO and RoboTwin 2.0 while demonstrating strong out-of-distribution generalization on LIBERO-Plus.",
    "abstract": "World Action Models (WAMs) couple robot action prediction with video world models. Existing WAMs with shared-backbone and Mixture-of-Transformers designs generally tie the depth of the action module to that of the video backbone, resulting in substantial computational overhead and high inference latency. To address this limitation, we introduce Dock of Transformer (DoT), a video-centric design principle that treats a pretrained video Transformer as a representation hub and connects lightweight output-heads through docking interfaces. This enables flexible output-head design while providing direct access to representations from all layers of the backbone. We then introduce \\textbf{Faster-WAM}, an instantiation of DoT for WAMs, which docks a single-layer action head onto a 30-layer video backbone. The docking interface fuses keys and values from all video layers and applies RoPE realignment. Without additional embodied pretraining, Faster-WAM achieves competitive performance on LIBERO and RoboTwin 2.0 while demonstrating strong out-of-distribution generalization on LIBERO-Plus. Faster-WAM also achieves the lowest end-to-end latency in our controlled comparison, requiring only 66.5 ms per inference --- a 3.2×3.2\\times speedup over Fast-WAM. Overall, these results demonstrate that the video-centric DoT architecture supports flexible task-specific head design while delivering low inference latency, strong action-prediction performance, and robust generalization.",
    "submittedDate": "2026-08-03",
    "primaryCategory": "3D/4D WAM",
    "secondaryCategories": [
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "FasterWAM",
    "arxivUrl": "https://arxiv.org/abs/2608.02365",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.02365",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "高效推理与实时控制"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.01397",
    "title": "SG-WAM: Self-Guided World Modeling in Geometry-Aware Policy Space",
    "authors": "Zhao, Ruiteng; Zhang, Zhengshen; Su, Yue; Wang, Wenshuo; Li, Jiahui; Yang, Zhiyuan; Tay, Francis E. H.; Jr., Marcelo H. Ang; Zhu, Haiyue",
    "affiliations": "1Advanced Robotics Centre, National University of Singapore,2MMLab, The University of Hongkong,3Nanyang Technological; University, 4Singapore Institute of Manufacturing Technology, Agency for Science, Technology and Research (A*STAR)",
    "contribution": "We propose SG-WAM, a self-guided framework that learns geometry-aware action-conditioned dynamics directly in the policy-derived representation space. Built on a 0.9B model without large-scale embodied pretraining, SG-WAM achieves 98.5% average success on LIBERO and 73% on LIBERO-Plus, while outperforming strong baselines in both in-distribution and out-of-distribution real-world evaluations.",
    "abstract": "World Action Models (WAMs) couple action generation with prediction of future states. Their effectiveness depends on whether future dynamics are modeled in a space that is both aligned with action generation and sufficiently geometry-aware to capture where and how actions change the scene. Existing WAMs typically satisfy only part of this requirement, relying on either perceptually heavy observation-space targets or auxiliary latent spaces that are not jointly structured for action relevance and geometry. We propose SG-WAM, a self-guided framework that learns geometry-aware action-conditioned dynamics directly in the policy-derived representation space. SG-WAM introduces learnable dynamics tokens and a Self-Guided World Predictor that forecasts their future latent states conditioned on intervening robot actions. Prediction targets are generated by an exponential moving average copy of the same policy backbone, providing stable supervision within the representation family used by the action expert. Geometric supervision further structures the policy image-token representations, providing spatially grounded context for the dynamics tokens and yielding a future-alignment space that is both action-relevant and geometry-aware. Latent future prediction, geometric grounding, and flow-matching action generation are jointly optimized end-to-end in a unified framework. Built on a 0.9B model without large-scale embodied pretraining, SG-WAM achieves 98.5% average success on LIBERO and 73% on LIBERO-Plus, while outperforming strong baselines in both in-distribution and out-of-distribution real-world evaluations.",
    "submittedDate": "2026-08-02",
    "primaryCategory": "3D/4D WAM",
    "secondaryCategories": [
      "Latent / Representation WAM"
    ],
    "bibtexKey": "SGWAM2",
    "arxivUrl": "https://arxiv.org/abs/2608.01397",
    "codeUrls": [],
    "projectUrl": "https://sg-wam.github.io/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.01397",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "潜空间预测与JEPA",
      "三维多视角建模",
      "泛化与动作对齐"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.01221",
    "title": "EndoWAM: A Grounded World-Action Model for Generalizable Endoscopic Navigation",
    "authors": "Lin, Jinsong; Pan, Zikang; Liu, Wanhao; Ng, Chi Kit; Shao, Liangjing; Yu, Zihang; Wang, Ziyu; Wang, Yin; Wang, Jiaxi; Teoh, Jeremy Yuen-Chun; Xiong, Zhiyong; Gao, Huxin; Ren, Hongliang",
    "affiliations": "Not identified",
    "contribution": "We present EndoWAM, which is, to our knowledge, the first WAM for generalizable robotic endoscopic navigation. EndoWAM consistently outperforms all baselines and alternative grounding strategies, while demonstrating strong zero-shot generalization to unseen viewpoints, environments, and targets.",
    "abstract": "Autonomous endoscopic navigation can reduce clinicians' operational burden, yet robust control remains challenging due to tissue deformation, transient occlusions, and rapidly changing viewpoints. Existing learning-based policies typically predict actions from current observations without explicitly modeling future dynamics, limiting their robustness and reliability in safety-critical settings. World Action Models (WAMs) offer a promising alternative by coupling predictive visual dynamics with action generation, but extending them to robotic endoscopy remains challenging due to limited training data, restricted viewpoint diversity, deformable anatomy, and high inference latency. We present EndoWAM, which is, to our knowledge, the first WAM for generalizable robotic endoscopic navigation. EndoWAM introduces future grounding, which predicts task-relevant target regions in future observations from intermediate denoising features of a video world model. Specifically, EndoWAM couples a lightweight diffusion transformer for future target-region prediction with a discrete action expert through a shared predictive representation. This design injects target-aware supervision into predictive dynamics modeling, improving robustness to visual degradation and viewpoint changes while enabling real-time control in a single denoising pass. We further introduce EndoMotion, a robotic endoscopic motion dataset spanning three anatomically distinct procedures: ureteroscopy, esophagoscopy, and endoscopic retrograde cholangiopancreatography (ERCP). EndoWAM consistently outperforms all baselines and alternative grounding strategies, while demonstrating strong zero-shot generalization to unseen viewpoints, environments, and targets. These results establish EndoWAM as a predictive, target-grounded framework for accurate, generalizable, and long-horizon navigation in visually constrained endoscopic environments.",
    "submittedDate": "2026-08-02",
    "primaryCategory": "Navigation / Driving / Domain WAM",
    "secondaryCategories": [
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "EndoWAM",
    "arxivUrl": "https://arxiv.org/abs/2608.01221",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.01221",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "导航",
      "视觉规划与IDM",
      "高效推理与实时控制"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.00725",
    "title": "SelfWAM: A Self-Grounded Unified World Action Model for Fast Robot Control",
    "authors": "Pan, Bikang; Liu, Fan; Lu, Haotao; Wang, Jingya; Shi, Ye",
    "affiliations": "1ShanghaiTech University 2InstAdapt",
    "contribution": "We introduce SelfWAM, a unified self-grounded WAM built on a modality-specialized Mixture-of-Transformers (MoT) architecture that jointly predicts actions, action-conditioned future RGB frames, and robot self-masks, thereby grounding future prediction in the robot's visible body and its action-induced motion.",
    "abstract": "World Action Models (WAMs) improve robot policy learning by jointly modeling actions and future observations. However, conditioning future prediction only on the task prompt and observation context risks capturing generic task progression rather than the action-specific consequences of the executed action. We introduce SelfWAM, a unified self-grounded WAM built on a modality-specialized Mixture-of-Transformers (MoT) architecture that jointly predicts actions, action-conditioned future RGB frames, and robot self-masks, thereby grounding future prediction in the robot's visible body and its action-induced motion. During joint training, SelfWAM allows future visual queries to attend to a clean copy of the demonstrated action, turning the video branch into an action-specific consequence model while leaving the fast action-only inference path unchanged. To focus video learning on action-relevant visual changes, we use prompt-specific objectives for future robot self-mask prediction, which removes appearance details and provides a target whose temporal evolution is tightly coupled with the conditioning action. Together, clean-action conditioning and future self-mask supervision make future predictions more directly reflect how the executed action changes the robot's visible motion and the surrounding scene. Experiments on RoboTwin 2.0 and real-world manipulation tasks show that SelfWAM produces more action-sensitive futures and preserves fast policy inference, while improving policy performance.",
    "submittedDate": "2026-08-01",
    "primaryCategory": "Real-Time / Efficient WAM",
    "secondaryCategories": [],
    "bibtexKey": "SelfWAM",
    "arxivUrl": "https://arxiv.org/abs/2608.00725",
    "codeUrls": [],
    "projectUrl": "https://selfwam.github.io/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.00725",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "泛化与动作对齐",
      "高效推理与实时控制"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.00635",
    "title": "FlowPilot: Real-Time World-Action Modeling for Agile UAV Navigation",
    "authors": "Wang, Runqing; Yu, Ding; Min, Pengyuan; Zhang, Xinhong; Xiao, Wei; Hu, Yu; Chen, Jie; Zhang, Fu; Wang, Gang",
    "affiliations": "Not identified",
    "contribution": "We present FlowPilot, a compact world-action model for real-time onboard UAV navigation from depth. In closed-loop simulation, it outperforms learning- and optimization-based baselines under increasing clutter and commanded speeds up to 8m/s.",
    "abstract": "We present FlowPilot, a compact world-action model for real-time onboard UAV navigation from depth. Unlike map-then-optimize pipelines that require local reconstruction or end-to-end policies that lack explicit scene prediction, FlowPilot jointly denoises future depth observations and executable trajectories with flow matching. A dual-stream mixture-of-transformers couples video and action experts through shared attention, allowing future-scene prediction and trajectory generation to inform each other. At deployment, the model runs action-centrically and outputs only a trajectory. To ensure trackability, actions are parameterized as degree-7 Bernstein polynomials: the current state constrains the initial control points, and the network predicts five free control points, yielding C^2-continuous references with closed-form velocity, acceleration and jerk. FlowPilot is trained on a three-level depth pyramid spanning high-throughput simulation, photorealistic simulation, and real onboard data. In closed-loop simulation, it outperforms learning- and optimization-based baselines under increasing clutter and commanded speeds up to 8m/s. On a physical quadrotor, the full perception-to-action pipeline runs in under 18ms on a Jetson Orin NX and reaches 5.5m/s in cluttered indoor and forest environments using only onboard sensing and computation.",
    "submittedDate": "2026-08-01",
    "primaryCategory": "Real-Time / Efficient WAM",
    "secondaryCategories": [
      "3D/4D WAM",
      "Navigation / Driving / Domain WAM"
    ],
    "bibtexKey": "FlowPilot",
    "arxivUrl": "https://arxiv.org/abs/2608.00635",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.00635",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "导航",
      "联合视频动作建模",
      "高效推理与实时控制"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "联合预测",
    "quadrant": "Q3 · Dual-system × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.00547",
    "title": "Disentangling Visuo-Tactile Foresight: Oracle-Guided Interface Discovery for World Action Models",
    "authors": "Yao, Zihang; Ding, Chaoyue; Yu, Yingying",
    "affiliations": "1 Brigham Young University, Provo, Utah, USA; 2 Beijing Academy of Science and Technology, Beijing, China; Corresponding author: Yingying Yu",
    "contribution": "To make this interface independently studyable, we introduce Oracle Visuo-Tactile Foresight (OVTF), a controlled framework that supplies paired RGB and tactile futures from successful trajectories verified in simulation. Within OVTF, we propose Asymmetric Phase-Local Future Memory (AFM), in which visual memory reads future vision, each tactile memory jointly attends to its own tactile stream and phase-aligned future vision, and cross-tactile access is blocked.",
    "abstract": "Contact-rich manipulation remains challenging because successful control depends on physical interaction cues that are often weakly observable from vision alone. Recent tactile world action models jointly model future visual observations and tactile signals to guide action generation, but how such futures should be structured for effective use by the action expert remains underexplored. Directly studying this question with learned world action models is difficult because end-to-end behavior entangles physically invalid visual futures, unreliable predictions, inaccurate or cross-modally inconsistent tactile forecasts, and an unreadable future-to-action interface. To make this interface independently studyable, we introduce Oracle Visuo-Tactile Foresight (OVTF), a controlled framework that supplies paired RGB and tactile futures from successful trajectories verified in simulation. By fixing the future provider, OVTF isolates the interface and asks a cleaner question: if the future is successful and physically executable, what representation allows the action expert to absorb its benefit? Within OVTF, we propose Asymmetric Phase-Local Future Memory (AFM), in which visual memory reads future vision, each tactile memory jointly attends to its own tactile stream and phase-aligned future vision, and cross-tactile access is blocked. We compare AFM with Modality-Isolated Future Memory (IFM), which removes visual-to-tactile access and processes each future modality independently. Across seven tasks on the UniVTAC simulation benchmark, AFM achieves 32.0% average success, compared with 23.7% for IFM and 14.9% for UniVTAC-ACT. This controlled comparison shows that selective phase-aligned visual-tactile routing provides a more actionable future-to-action bridge than complete modality isolation.",
    "submittedDate": "2026-08-01",
    "primaryCategory": "Multimodal / Tactile WAM",
    "secondaryCategories": [
      "Evaluation / Survey / Theory",
      "Memory WAM"
    ],
    "bibtexKey": "OVTF",
    "arxivUrl": "https://arxiv.org/abs/2608.00547",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.00547",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "评估指标与协议",
      "动作策略基础"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.14655",
    "title": "Diagnosing and Mitigating Perception-Decision Misalignment in Omni-LLMs via Modality Subspace Activation",
    "authors": "Jiang, Hongbo; Li, Jie; Shen, Yunhang; Xie, Tianyu; Dai, Pingyang",
    "affiliations": "1Xiamen University; 2Shanghai Artificial Intelligence Laboratory; 3Tencent YouTu Laboratory",
    "contribution": "To rectify this, we propose Modality Subspace Activation (MSA), a training-free inference-time framework that uses Singular Value Decomposition (SVD) to estimate modal activation strengths.",
    "abstract": "Omni-Large Language Models (Omni-LLMs) power complex multi-modal reasoning in applications like World Action Models and autonomous agents. However, their strong performance often masks a profound Perceptual-Decision Misalignment (PDM), where decisions remain unfaithful to multi-modal perceptions. To diagnose this, we formalize Causal Modality Sensitivity (CMS), operationalized via a dual-lens framework: Answer Retention Rate (ARR) at the macro behavioral level, and Logit Angular Discrepancy (LAD) to track microscopic distribution shifts. We also curate CausalMSBench, a diagnostic dataset isolating language priors. Benchmarking reveals that popular Omni-LLMs exhibit critically low CMS, showing negligible distribution shifts even when key modalities are removed. To rectify this, we propose Modality Subspace Activation (MSA), a training-free inference-time framework that uses Singular Value Decomposition (SVD) to estimate modal activation strengths. MSA dynamically balances modal projections in the last hidden state, effectively restoring CMS across benchmarks.",
    "submittedDate": "2026-07-31",
    "primaryCategory": "Multimodal / Tactile WAM",
    "secondaryCategories": [
      "Evaluation / Survey / Theory"
    ],
    "bibtexKey": "OmniLLMs",
    "arxivUrl": "https://arxiv.org/abs/2608.14655",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.14655",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": null,
    "subcategories": [],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2607.29235",
    "title": "FBFM: A Training-Free Asynchronous Feedback Mechanism for Flow-Matching in World-Action Models Execution",
    "authors": "Li, Peize; Zhang, Ruimeng; Zhang, Ru; Huang, Cong; Chen, Kai; Zhang, Shanghang",
    "affiliations": "1DeepCybo 2Zhongguancun Academy 3Zhongguancun Institute of Artificial Intelligence; 4Peking University 5Beijing Institute of Technology 6Tsinghua University",
    "contribution": "To address this, we propose Feedback Flow Matching (FBFM), a training-free inference mechanism that pushes re-grounding inside the actively generated chunk.",
    "abstract": "Although world-action models (WAMs) enhance long-horizon robot control by predicting visual evolution before acting, long-horizon reliability demands repeated re-grounding in real observations--not recursive rollout. Existing WAMs address this by refreshing history or KV cache with ground-truth data between chunks. However, such chunk-wise feedback operates at a coarse temporal granularity and thus fails to correct prediction errors at the individual time-step level. To address this, we propose Feedback Flow Matching (FBFM), a training-free inference mechanism that pushes re-grounding inside the actively generated chunk. During flow matching, FBFM applies a masked pseudoinverse correction to the conditional velocity field: it leverages the preceding action chunk to guide generation of the next action chunk, and uses the image observed after executing that preceding chunk to guide the next frame prediction. This cross-chunk pairing--where feedback from one chunk arrives in time to shape the next--creates an asynchronous loop that corrects errors without waiting for chunk boundaries. Being training-free, the mechanism improves responsiveness to unexpected events and suppresses drift in long-horizon tasks. We evaluate FBFM on both a joint-generation WAM (DreamZero) and a stage-wise WAM (LingBot-VA). On selected LIBERO and RoboTwin2.0 tasks, it improves success rates by over 5% in favorable settings, and real-world robot observation-prediction diagnostics show notably better tracking. We argue that FBFM offers a new paradigm for fine-grained online correction, bridging open-loop flow generation with closed-loop real-world dynamics.",
    "submittedDate": "2026-07-31",
    "primaryCategory": "Real-Time / Efficient WAM",
    "secondaryCategories": [],
    "bibtexKey": "FBFM",
    "arxivUrl": "https://arxiv.org/abs/2607.29235",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2607.29235",
    "pdfUrl": "https://arxiv.org/pdf/2607.29235",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{FBFM,\n  title={FBFM: A Training-Free Asynchronous Feedback Mechanism for Flow-Matching in World-Action Models Execution},\n  author={Li, Peize and Zhang, Ruimeng and Zhang, Ru and Huang, Cong and Chen, Kai and Zhang, Shanghang},\n  journal={arXiv:2607.29235},\n  year={2026}\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "高效推理与实时控制",
      "记忆与长时序"
    ],
    "architecture": "不适用",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2608.14650",
    "title": "Paired Exact-Reset Evaluation of a Prediction-Derived Medium-to-Full World-Model Cascade",
    "authors": "Pastor, Malo de",
    "affiliations": "Not identified",
    "contribution": "Existing adaptive-inference and world-action-model systems use cheap-stage outputs or predicted futures to allocate additional computation. We study a narrower question: under paired exact-reset physical outcomes, can a Medium-derived interface predict when switching to a separately frozen Full predictor improves task-specific decision loss enough to justify sequential overhead?",
    "abstract": "Existing adaptive-inference and world-action-model systems use cheap-stage outputs or predicted futures to allocate additional computation. We study a narrower question: under paired exact-reset physical outcomes, can a Medium-derived interface predict when switching to a separately frozen Full predictor improves task-specific decision loss enough to justify sequential overhead? Our contribution is a paired evaluation and audit protocol, not a new generic routing rule: all candidate actions are executed from the same reset state, Medium and Full act on the same candidate set and task, and their paired physical-loss difference defines the routing target. On a fresh PushT bank (V106; 1,600 states, 39 tasks, three checkpoint pairs), a frozen prediction-interface router lowers overhead-inclusive decision cost relative to standalone Medium, standalone Full, and a latency-advantaged task-only router. We then prospectively seal a second 1,600-state PushT confirmation (V107) against a stronger current-state control using the task, a dimension-matched projection of current DINO features, and all five candidate actions, with no DINO encoder latency charged. The prediction interface lowers priced physical decision cost by 0.002549 (state-clustered 95% interval [-0.002867, -0.002238]; one-sided 95% upper bound -0.002286), with negative effects for all three checkpoint pairs. A controlled-PyBullet audit independently supports a composite task-prediction-regime router. The sequential router remains slower than fixed policies, and its advantage is restricted to low compute prices. The evidence supports incremental routing information in the tested prediction interface beyond one deliberately favoured current-DINO control, but not causal sufficiency, compute saving, closed-loop value, or cross-family generality.",
    "submittedDate": "2026-07-30",
    "primaryCategory": "Evaluation / Survey / Theory",
    "secondaryCategories": [],
    "bibtexKey": "V107",
    "arxivUrl": "https://arxiv.org/abs/2608.14650",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2608.14650",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "评估指标（Metrics）",
    "subcategories": [
      "统计评测协议",
      "评估指标与协议"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2607.28993",
    "title": "ST-WAM: Semantic-Temporal World Action Model for Robust Manipulation under Visual Distribution Shifts",
    "authors": "Wang, Mingxin; Hu, Bin; Qian, Bin; Jiang, Kaitao; Wu, Haoning; Yan, Feng; Jing, Bowen; Hao, Ruiyang; Wang, Enyi; Niu, Kangning; Yang, Yandan; Xu, Mu; Wang, Yan; Liu, Houde; Li, Tianlun",
    "affiliations": "1Tsinghua University, 2AMAP-CV-LAB, Alibaba Group; 3Shanghai Jiao Tong University,4Xi’an Jiaotong University; 5The University of Manchester,6King’s College London",
    "contribution": "Rather than correcting the predicted futures, we propose Semantic-Temporal WAM (ST-WAM) to improve action robustness by using DINOv3 as a shared semantic representation for future prediction and history retrieval while retaining fine-grained VAE dynamics. It achieves 98.7% on LIBERO and 92.8% on RoboTwin 2.0; more importantly, compared with Fast-WAM, it improves zero-shot LIBERO-Plus performance by 21.3 percentage points and more than doubles real-world success under visual shifts from 25.8% to 61.5%.",
    "abstract": "World Action Models (WAMs) have emerged as a promising paradigm by jointly modeling robot actions and future visual dynamics. However, their reliance on pixel-generative future supervision can entangle action-relevant state transitions with task-irrelevant visual content, limiting robustness under visual distribution shifts. We identify Training-Distribution Hallucination, a recurring phenomenon in which futures conditioned on visually shifted observations hallucinate training-domain content rather than remain faithful to the current scene. A controlled frame-triplet diagnosis further shows that DINOv3 features remain more stable across visual shifts while better preserving task-state distinctions than Wan-VAE latents. Rather than correcting the predicted futures, we propose Semantic-Temporal WAM (ST-WAM) to improve action robustness by using DINOv3 as a shared semantic representation for future prediction and history retrieval while retaining fine-grained VAE dynamics. Its Dual-Space Future Experts (DSFE) jointly predict future VAE latents and DINO features, while Current-Anchored Intent Retrieval (CAIR) retrieves task-relevant evidence from recent DINO history under the current visual-language context. ST-WAM is trained end-to-end without additional embodied pretraining or task-specific annotations, and requires no explicit future generation at inference. It achieves 98.7% on LIBERO and 92.8% on RoboTwin 2.0; more importantly, compared with Fast-WAM, it improves zero-shot LIBERO-Plus performance by 21.3 percentage points and more than doubles real-world success under visual shifts from 25.8% to 61.5%. These results demonstrate that semantic-temporal modeling effectively complements pixel-generative dynamics for robust manipulation.",
    "submittedDate": "2026-07-30",
    "primaryCategory": "Memory WAM",
    "secondaryCategories": [
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "STWAM",
    "arxivUrl": "https://arxiv.org/abs/2607.28993",
    "codeUrls": [
      "https://github.com/Thu-WangMX/ST-WAM-Semantic-Temporal-World-Action-Model"
    ],
    "projectUrl": "https://thu-wangmx.github.io/st-wam/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2607.28993",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "潜空间预测与JEPA",
      "记忆与长时序",
      "高效推理与实时控制"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2607.28405",
    "title": "QuantWAMs: Calibrating at the Right Granularity for World Action Models",
    "authors": "Zhou, Jiacheng; Lv, Jinfan; Li, Ruixuan; Zhang, Longtai; Wang, Yan; Zhang, Wenqiang; Qi, Lizhe",
    "affiliations": "College of Intelligent Robotics and Advanced Manufacturing, Fudan University; Shanghai Key Lab of Intelligent Information Processing; College of Computer Science and Artificial Intelligence, Fudan University; School of Data Science and Engineering, East China Normal University",
    "contribution": "We present QuantWAMs, a PTQ framework that aligns quantization decisions with the calibration context defined by model structure, rollout distribution, and task objective.",
    "abstract": "World Action Models (WAMs) jointly predict future observations and actions, but their iterative denoising and closed-loop execution make efficient deployment costly. Existing post-training quantization (PTQ) methods are poorly suited to WAMs because they rely on open-loop objectives, homogeneous model assumptions, and calibration distributions that do not reflect deployment. We present QuantWAMs, a PTQ framework that aligns quantization decisions with the calibration context defined by model structure, rollout distribution, and task objective. QuantWAMs introduces three strategies: shared-basis outlier calibration, which pools activation evidence only across coordinate-compatible modules; co-training-objective saliency, which computes empirical-Fisher scores from the joint video--action gradient and assigns weight precision at a calibration-stable layer granularity; and fixed-intervention rollout auditing, which revises denoising-step protection schedules using reachable closed-loop states without changing the precision budget. We evaluate QuantWAMs on Fast-WAM and LingBot-VA across RoboTwin 2.0, LIBERO, and real-robot manipulation with an AgiBot G2. Under a W4A4-dominant setting, the reported simulation means differ from FP16 by 0.2--0.7 percentage points. Real-robot trials further establish deployment feasibility on three manipulation tasks. For the targeted video and action blocks, QuantWAMs reduces peak weight-and-activation memory to about 29\\% of FP16 and provides 1.4--1.6×\\times block-level speedups.",
    "submittedDate": "2026-07-30",
    "primaryCategory": "Memory WAM",
    "secondaryCategories": [
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "QuantWAMs",
    "arxivUrl": "https://arxiv.org/abs/2607.28405",
    "codeUrls": [],
    "projectUrl": "https://quantwams.github.io/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2607.28405",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "高效推理与实时控制"
    ],
    "architecture": "不适用",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2607.28243",
    "title": "EgoGenesis: Egocentric World-Action Modeling with Online Anchored Projective Memory and Action-3D RoPE",
    "authors": "Yan, Zexuan; Wu, Yuzhou; Ma, Yue; He, Zonghang; Yin, Kaibo; Tu, Xiaobing; Wang, Yinggui; Ren, Jinkui; Zhang, Xiantao; Wang, Shijian; Liu, Jinghong; Zhang, Linfeng",
    "affiliations": "1 Shanghai Jiao Tong University; 2 Alibaba Group; 3 Tianji KernalMind Co., Ltd; 4 The Hong Kong University of Science and Technology; 5 Southeast University; 6 Renmin University of China; 7 The University of Tokyo",
    "contribution": "We present \\method, an egocentric world-action simulator that synthesizes controllable, high-quality manipulation videos to expand scarce real-world training data. \\method{} builds on a pretrained video generation prior and introduces two geometry-aware conditioning mechanisms.",
    "abstract": "Egocentric video offers rich manipulation experience for embodied AI, yet collecting diverse egocentric data across scenes, objects, motions, and embodiments remains costly. We present \\method, an egocentric world-action simulator that synthesizes controllable, high-quality manipulation videos to expand scarce real-world training data. \\method{} builds on a pretrained video generation prior and introduces two geometry-aware conditioning mechanisms. Online Anchored Projective Memory (OAPM) preserves a first-frame 3D scene anchor while periodically refreshing a recent state during autoregressive generation. Action-3D Rotary Position Embedding (A3D-RoPE) encodes end-effector motion with camera-aware 3D rotary coordinates, injecting action geometry into skeleton-to-video cross-attention for precise control. Together, these components improve visual fidelity, geometric stability, and action alignment in long egocentric rollouts. Moreover, augmenting 400 real trajectories with 400 \\method-generated trajectories improves out-of-distribution real-robot success from 77\\% to 84\\% on single-arm tasks and from 53\\% to 70\\% on dual-arm tasks, demonstrating that the synthesized data substantially improve downstream WAM generalization.",
    "submittedDate": "2026-07-30",
    "primaryCategory": "3D/4D WAM",
    "secondaryCategories": [
      "Latent / Representation WAM",
      "Memory WAM"
    ],
    "bibtexKey": "EgoGenesis",
    "arxivUrl": "https://arxiv.org/abs/2607.28243",
    "codeUrls": [],
    "projectUrl": "https://egogenesis.github.io/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2607.28243",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "神经世界模拟器"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2607.23969",
    "title": "LeapBot-WA: World-Anchor Action Models via Predictive Latent Alignments",
    "authors": "Liu, Pei; Zheng, Nan; Zhang, Lang; Peng, Daojie; Zhang, Yanan; Kong, Feilong; Feng, Mingyue; Liu, Jiachao; Wang, Yaonong; Chen, Qifeng; Ma, Jun",
    "affiliations": "1The Hong Kong University of Science and Technology (Guangzhou); 2The Hong Kong University of Science and Technology; 4Southeast University",
    "contribution": "In this paper, we propose LeapBot-WA, which establishes a novel Predictive-Latent paradigm for WAMs by operationalizing the Joint-Embedding Predictive Architecture (JEPA) as a World-Anchor. To bridge the modality gap between non-Gaussian predictive features and diffusion priors, we introduce the Isotropic Semantic Autoencoder (ISAE), which reshapes the anchor's latent space into a diffusion-friendly manifold to prevent off-manifold drift.",
    "abstract": "World Action Models (WAMs) have emerged as a powerful paradigm for embodied intelligence, yet the prevailing reliance on pixel-level video generation creates a fundamental bottleneck. Forcing models to reconstruct task-irrelevant visual details dissipates representational capacity and renders policies vulnerable to visual distractors. In this paper, we propose LeapBot-WA, which establishes a novel Predictive-Latent paradigm for WAMs by operationalizing the Joint-Embedding Predictive Architecture (JEPA) as a World-Anchor. Departing from the traditional reliance on visual synthesis, LeapBot-WA shifts the core of world modeling to Predictive Semantic Alignment, extracting abstract physical dynamics directly within a latent foundation space. To bridge the modality gap between non-Gaussian predictive features and diffusion priors, we introduce the Isotropic Semantic Autoencoder (ISAE), which reshapes the anchor's latent space into a diffusion-friendly manifold to prevent off-manifold drift. Furthermore, we design an Asymmetric Mixture-of-Transformers (MoT) architecture. During training, an Anchor Diffusion Transformer acts as a privileged dynamics expert to guide the Action Diffusion Transformer; at inference, this heavy dynamics branch is pruned, enabling zero-overhead execution. LeapBot-WA achieves state-of-the-art performance among predictive models on LIBERO and matches top-tier generative WAMs on RoboTwin 2.0 without requiring large-scale trajectory pre-training. It further demonstrates superior zero-shot robustness to unseen environments and successful real-world transfer, establishing a highly efficient and robust latent-centric paradigm for scalable robotic control. Code: https://github.com/LeapWM/leapbot-wa.",
    "submittedDate": "2026-07-30",
    "primaryCategory": "Latent / Representation WAM",
    "secondaryCategories": [
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "LeapBotWA",
    "arxivUrl": "https://arxiv.org/abs/2607.23969",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2607.23969",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "潜空间预测与JEPA",
      "高效推理与实时控制"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2607.25918",
    "title": "DC-WAM: Dynamic-Centric Visual Supervision and Reasoning for World-Action Models",
    "authors": "Ji, Haoyuan; Fan, Lingxiang; Su, Shang; Lu, Yinqiao; Shi, Mengkai; Gao, Jun; Feng, Shuo",
    "affiliations": "1Tsinghua University",
    "contribution": "We propose DC-WAM, a dynamic-centric WAM framework that redistributes supervision and computation in the RGB video branch.",
    "abstract": "World-Action Models (WAMs) augment robot policies with future visual prediction, but it remains unclear what the visual modality should learn for control. While photorealistic future prediction provides dense supervision, it also incurs substantial computation and can allocate capacity to texture, illumination, and background variations that are only weakly related to action selection. Recent efficient WAM variants suggest that the main benefit of the video branch may not lie in the rendered future itself, but in the control-relevant visual representations induced during training. In this work, we revisit future video prediction from a dynamic-centric perspective and ask whether an existing RGB-based WAM can be redirected from appearance-dominated reconstruction toward interaction-induced visual dynamics without introducing additional modality-specific predictions or online inputs at deployment. We propose DC-WAM, a dynamic-centric WAM framework that redistributes supervision and computation in the RGB video branch. At the supervision level, DC-WAM combines temporal-difference flow matching with trajectory-guided weighting, emphasizing dense temporal changes and localized regions where the gripper, manipulated objects, and contact areas move. At the reasoning level, DynaRoute predicts token-wise dynamic relevance and converts it into an attention bias, guiding the model toward control-relevant future tokens. Experiments in simulation and on real-world manipulation tasks show that DC-WAM consistently improves policy performance, especially under out-of-distribution perturbations in lighting, object appearance, and background texture.",
    "submittedDate": "2026-07-28",
    "primaryCategory": "Real-Time / Efficient WAM",
    "secondaryCategories": [],
    "bibtexKey": "DCWAM",
    "arxivUrl": "https://arxiv.org/abs/2607.25918",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2607.25918",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "联合视频动作建模",
      "泛化与动作对齐"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "联合预测",
    "quadrant": "Q3 · Dual-system × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2607.25895",
    "title": "HiFi-UMI: Learning Deployable Manipulation Policies from High-Fidelity UMI Data Alone",
    "authors": "AI, Simple; :; Wei, Yuteng; Ma, Jinming; Wang, Jiawei; Zhou, Weitao; Zuo, Yushen; Rui, Ke; Li, Minglei; Zhang, Jinhao; Pan, Zhikang; Wang, Xiang; Jia, Haoran; Du, Huan; Zeng, Zicheng; Ma, Jun; Qin, Guiyu; Zhang, Di; Li, Xiaofei",
    "affiliations": "Website: Dataset:",
    "contribution": "We present HiFi-UMI, a portable UMI data-production system co-designed for trajectory accuracy, inter-gripper relative pose, synchronization, and field of view: head-mounted offline stereo-inertial SLAM, native rather than reconstructed relative pose, a shared microsecond GPIO trigger, and two wide-angle cameras per hand covering ~200 degrees.",
    "abstract": "Learning deployable manipulation policies is bottlenecked by the scarcity of data that is both high-fidelity and scalable. Real-robot teleoperation is accurate but costly to scale; robot-free UMI capture scales readily, and current practice uses the resulting data mainly for pre-training, adding a small real-robot \"anchor\" at post-training. We ask whether raising the fidelity of robot-free UMI data, rather than shrinking the real-robot fraction, can remove that anchor. We present HiFi-UMI, a portable UMI data-production system co-designed for trajectory accuracy, inter-gripper relative pose, synchronization, and field of view: head-mounted offline stereo-inertial SLAM, native rather than reconstructed relative pose, a shared microsecond GPIO trigger, and two wide-angle cameras per hand covering ~200 degrees. It reaches 3 mm workspace-local end-effector accuracy without external tracking infrastructure. Using this corpus, we demonstrate zero-robot post-training: a policy post-trained solely on HiFi-UMI demonstrations deploys directly on a real robot and matches in-domain teleoperation across three backbones spanning the vision-language-action and world-action-model families, with success-rate differences of -2.5, +3.1, and -0.6 percentage points on StarVLA-QwenPI, OpenPI-pi_0.5, and LingBot-VA; the strongest policy reaches 85% on a precision insertion task, even though the teleoperation baseline is collected in the evaluation scene and no HiFi-UMI trajectory is. Pre-training on 4,000 hours from the same corpus lowers action error on ten unseen tasks by 41% and, on StarVLA-QwenPI, raises real-robot success by a further 18.1 percentage points. We open-source HiFi-UMI-2K, 2,000 hours of microsecond-synchronized, ultra-wide-FoV demonstrations, each automatically reconstructed and validated through simulation replay, as a large-scale, high-fidelity resource for the robot-learning community.",
    "submittedDate": "2026-07-28",
    "primaryCategory": "Evaluation / Survey / Theory",
    "secondaryCategories": [],
    "bibtexKey": "HiFiUMI",
    "arxivUrl": "https://arxiv.org/abs/2607.25895",
    "codeUrls": [],
    "projectUrl": "https://cloud.simpleai.tech/simple-world-lab/hifi-umi/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2607.25895",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "数据集",
    "subcategories": [
      "数据采集接口",
      "机器人示范与操作数据",
      "人类第一视角数据"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2607.18060",
    "title": "RoboHarness: Memory-Driven Orchestration of Heterogeneous Robot Policies for Long-Horizon Planning",
    "authors": "Huang, Jinbang; Hu, Yuanzhao; Li, Zhiyuan; Qi, Ran; Xiao, Yixin; Zhang, Zhanguang; Coates, Mark; Cao, Tongtong; Zhang, Yingxue",
    "affiliations": "Huawei Noah's Ark Lab; University of British Columbia; University of Toronto; McGill University，; Department of Foundation Model, 2012 Labs; Work done during the intership at Huawei Noah's Ark Lab",
    "contribution": "We propose RoboHarness, a unified framework that encapsulates independently developed robot control systems as reusable agentic skills.",
    "abstract": "Long-horizon robotic tasks require diverse capabilities that no single policy can reliably provide. Heterogeneous policies offer complementary strengths, but orchestrating them requires reasoning over uncertain capability boundaries and cross-policy distribution mismatch, which are largely overlooked by existing planning methods built on homogeneous, predefined skills with fixed applicability. We propose RoboHarness, a unified framework that encapsulates independently developed robot control systems as reusable agentic skills. Although instantiated in this work with VLAs, RL policies, and task-and-motion planning (TAMP) systems, RoboHarness is designed as a general framework compatible with a broader range of robot policies, such as navigation policies, model predictive controllers, and world-action models. RoboHarness uses multi-modal execution memory and online evidence to characterize policy capability boundaries for capability-aware decomposition and routing. To stabilize policy handoffs, its Memory Bridge retrieves execution trajectories associated with the next policy, estimates its in-distribution state region, and guides the robot toward that region without joint policy retraining. Extensive experiments on three public benchmarks, 500 customized tasks, and 135 real-robot experiments demonstrate effective capability-aware routing and stable policy orchestration, yielding substantial improvements in zero-shot long-horizon planning and out-of-distribution robustness.",
    "submittedDate": "2026-07-28",
    "primaryCategory": "Memory WAM",
    "secondaryCategories": [
      "Multimodal / Tactile WAM",
      "Navigation / Driving / Domain WAM"
    ],
    "bibtexKey": "RoboHarness",
    "arxivUrl": "https://arxiv.org/abs/2607.18060",
    "codeUrls": [],
    "projectUrl": "https://www.robo-harness.com/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2607.18060",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "Related resources",
    "subcategories": [
      "理论与规划",
      "综述与技术资源"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2607.23783",
    "title": "N₀-TWAM: Scaling Tactile-Native World-Action Model for Contact-Rich Manipulation",
    "authors": "Team, NeoteAI; Team, Fudan TEAI",
    "affiliations": "Not identified",
    "contribution": "We present N0N_0-TWAM, a tactile-native world-action model for contact-rich manipulation that predicts both future vision and future contact. To improve long-horizon and multi-stage manipulation, we introduce tactile contact events for task staging and advance through them during execution.",
    "abstract": "We present N0N_0-TWAM, a tactile-native world-action model for contact-rich manipulation that predicts both future vision and future contact. To our knowledge, it is the first tactile world-action model trained at large scale, and it shows strong capability on contact-rich tasks. We pre-train N0N_0-TWAM at large scale with visuo-tactile joint training over tactile-rich demonstrations spanning six embodiments and 450 tasks. We use NeoForce, a unified force-based tactile representation, to form a physically grounded contact signal that conditions action generation. To improve long-horizon and multi-stage manipulation, we introduce tactile contact events for task staging and advance through them during execution. For real-time efficiency, we adopt an asymmetric Mixture-of-Transformers architecture that pairs a full-width expert for video prediction with slim experts for downstream action and tactile prediction. Evaluations on both real and simulated benchmarks justify the capabilities of N0N_0-TWAM across a range of contact-rich tasks, and demonstrate the benefit of data scaling for precise tactile and action prediction. In summary, N0N_0-TWAM endows a world-action model with predictive capabilities to foresee vision, touch and action, building a solid foundation for fine-grained manipulation on open contact-rich tasks. The codebase and model checkpoints will be made publicly available to foster further research and development in tactile-enabled robotic manipulation.",
    "submittedDate": "2026-07-26",
    "primaryCategory": "Multimodal / Tactile WAM",
    "secondaryCategories": [
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "N0TWAM",
    "arxivUrl": "https://arxiv.org/abs/2607.23783",
    "codeUrls": [
      "https://github.com/neoteai/N0-TWAM"
    ],
    "projectUrl": "https://research.neoteai.com/n0-twam/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2607.23783",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "多模态触觉音频",
      "视觉规划与IDM",
      "记忆与长时序"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2607.17521",
    "title": "GeoWorldAD: Geometry World Action Model for Autonomous Driving",
    "authors": "Zhang, Songyan; Tian, Jinyuan; Li, Hanbing; Liu, Daqi; Chen, Hao; Huang, Wenhui; Li, Fang; Chen, Guang; Ye, Hangjun; Chen, Long; Yang, Kuiyuan; Lv, Chen",
    "affiliations": "1 Nanyang Technological University, 2 Xiaomi EV, 3 Zhejiang University",
    "contribution": "In this work, we propose GeoWorldAD, a geometry world action model that grounds trajectory planning in ego-aligned 3D space and anticipates short-horizon scene evolution with latent future geometry tokens. Experiments on NAVSIM v1 and v2 demonstrate state-of-the-art performance, highlighting the effectiveness of explicit 3D geometry grounding and future geometry world modeling for safe and efficient autonomous driving.",
    "abstract": "Autonomous driving requires both safe and efficient planning decisions in dynamic 3D environments. Although recent Vision/Video-Action models learn policies directly from visual observations and scale well with advances in vision transformers and large-scale training data, they often lack explicit geometric grounding and future-aware spatial guidance, limiting their ability to balance collision avoidance and driving progress. In this work, we propose GeoWorldAD, a geometry world action model that grounds trajectory planning in ego-aligned 3D space and anticipates short-horizon scene evolution with latent future geometry tokens. Present geometry provides essential spatial constraints for safe planning, while future geometry reveals how surrounding agents and ego-centric free space may evolve, reducing overly conservative decisions without sacrificing safety. To efficiently exploit these geometric cues, GeoWorldAD progressively aggregates multi-scale present geometry and latent future geometry through iterative trajectory refinement. Experiments on NAVSIM v1 and v2 demonstrate state-of-the-art performance, highlighting the effectiveness of explicit 3D geometry grounding and future geometry world modeling for safe and efficient autonomous driving.",
    "submittedDate": "2026-07-23",
    "primaryCategory": "3D/4D WAM",
    "secondaryCategories": [
      "Latent / Representation WAM",
      "Navigation / Driving / Domain WAM",
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "GeoWorldAD",
    "arxivUrl": "https://arxiv.org/abs/2607.17521",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2607.17521",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "三维多视角建模",
      "自动驾驶",
      "视觉规划与IDM"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2606.03159",
    "title": "NVIDIA OmniDreams: Real-Time Generative World Model for Closed-Loop Autonomous Vehicle Simulation",
    "authors": "NVIDIA; Aarti Basant; Amlan Kar; Despoina Paschalidou; Fangyin Wei; Francesco Ferroni; Guillermo Garcia Cobo; Haithem Turki; Huan Ling; Jaewoo Seo; James Lucas; Jay Zhangjie Wu; Jialiang Wang; Jonathan Lorraine; Jun Gao; Kai He; Katarina Tothova; Kevin Xie; Michał Tyszkiewicz; Qi Wu; Riccardo de Lutio; Ruilong Li; Sanja Fidler; Seung Wook Kim; Tianchang Shen; Tianshi Cao; Tobias Pfaff; William Lew; Xindi Wu; Xuanchi Ren; Yifan Lu; Yuxuan Zhang; Zan Gojcic; Zian Wang",
    "affiliations": "NVIDIA A detailed list of contributors and acknowledgments can be found in sec::contributors of this paper",
    "contribution": "To overcome these limitations, we introduce OmniDreams, a foundation generative world model mid- and post-trained from the Cosmos diffusion model to autoregressively generate action-conditioned videos in real time. We additionally show preliminary results indicating that a world-action model (WAM) post-trained from OmniDreams achieves strong performance on the Physical AI Autonomous Vehicles NuRec dataset, surpassing the VLA-based Alpamayo 1.5 research policy model while using only 1/5 the total parameters.",
    "abstract": "As autonomous vehicle capabilities advance, the safe evaluation of driving policies in long-tail scenarios remains a critical bottleneck. In closed-loop simulation, the driving policy model actively interacts with the environment, where its actions dynamically update the simulator state and directly influence the next set of generated sensor observations. While recent reconstruction-based neural simulators offer photorealism, they are fundamentally constrained by their initial captured data and struggle to generalize to highly dynamic or novel scenes. To overcome these limitations, we introduce OmniDreams, a foundation generative world model mid- and post-trained from the Cosmos diffusion model to autoregressively generate action-conditioned videos in real time. By leveraging the rich visual priors of Cosmos and mid- and post-training on 21k hours of driving scenarios, OmniDreams synthesizes complex, unobserved phenomena that are hard for traditional simulators to capture, such as extreme weather and unpredictable dynamic agent behaviors. Crucially, it autoregressively conditions its photorealistic sensor generation on past frames, the current simulator state, and immediate driving actions. Deployed in a closed-loop system with the Alpamayo 1 policy model and AlpaSim orchestrator, OmniDreams acts as a highly responsive, reactive environment, providing a scalable and comprehensive solution for training and evaluating next-generation autonomous driving policies. We additionally show preliminary results indicating that a world-action model (WAM) post-trained from OmniDreams achieves strong performance on the Physical AI Autonomous Vehicles NuRec dataset, surpassing the VLA-based Alpamayo 1.5 research policy model while using only 1/5 the total parameters. These results highlight the potential for a real-time world model like OmniDreams to also serve as a backbone for policy architectures.",
    "submittedDate": "2026-07-23",
    "primaryCategory": "Real-Time / Efficient WAM",
    "secondaryCategories": [
      "Evaluation / Survey / Theory",
      "Navigation / Driving / Domain WAM"
    ],
    "bibtexKey": "NVIDIAOmniDreams",
    "arxivUrl": "https://arxiv.org/abs/2606.03159",
    "codeUrls": [
      "https://github.com/nv-tlabs/omni-dreams"
    ],
    "projectUrl": "https://research.nvidia.com/labs/sil/projects/omnidreams-blog/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2606.03159",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "神经世界模拟器"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2607.20175",
    "title": "PerceptDrive: Perception Prior World-Action Modeling with Adaptive Expert Routing for End-to-End Autonomous Driving",
    "authors": "Liu, Yushan; Lv, Tianxiong; Wang, Bohua; Fan, Hangqi; Zhao, Chenxu; Zheng, He; Zhong, Xuchang; Xie, Yifan; Zhao, Congyang; Liao, Zhihao; Luo, Leigang; Cai, Yang; Zhang, Xiao-Ping; Ding, Wenbo",
    "affiliations": "1Tsinghua University 2AMap, Alibaba Group; ∗Work done during internship at AMap, Alibaba Group",
    "contribution": "Experiments show that PerceptDrive achieves state-of-the-art performance with 90.4 PDMS on NAVSIM v1 and 90.2 EPDMS on NAVSIM v2, outperforming existing methods.",
    "abstract": "Frozen perception foundation models encode rich geometric, semantic, and dynamic knowledge. Yet narrow conditioning interfaces may attenuate task-relevant cues, while static fusion cannot adjust expert contributions to each scene. We cast this challenge as the prior-to-plan transfer problem and introduce PerceptDrive, a perception prior world-action modeling framework with adaptive expert routing. PerceptDrive feeds teacher-distilled priors from a frozen, driving-adapted provider and dense observation latents from a frozen self-supervised video encoder into a trainable expert-routed world-action model. Expert-specific query branches process these signals, while a prior-retention objective anchors each branch to its prior. A router predicts soft gates from a shared scene representation and combines the expert conditions before trajectory generation. During training, privileged rule-based sub-metric estimates for branch-specific trajectory drafts provide soft-gate distillation targets. The predicted action-free future latent conditions a flow-matching actor. At inference, privileged components are absent; with one front-facing camera, PerceptDrive generates one trajectory per planning step without test-time scoring, reranking, or search. Experiments show that PerceptDrive achieves state-of-the-art performance with 90.4 PDMS on NAVSIM v1 and 90.2 EPDMS on NAVSIM v2, outperforming existing methods. Ablations confirm complementary gains from prior retention and scene-conditioned routing, alongside differential reliance on the three priors. These results demonstrate that preserving and adaptively routing perception priors improves direct planning without test-time candidate selection.",
    "submittedDate": "2026-07-22",
    "primaryCategory": "Navigation / Driving / Domain WAM",
    "secondaryCategories": [
      "3D/4D WAM",
      "Latent / Representation WAM",
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "PerceptDrive",
    "arxivUrl": "https://arxiv.org/abs/2607.20175",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2607.20175",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "自动驾驶",
      "潜空间预测与JEPA",
      "视觉规划与IDM"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2607.18840",
    "title": "WorldScape Policy 2.0: Empowering Steerable World Action Modeling with Reasoning-Augmented Memory",
    "authors": "Su, Haisheng; Liu, Zongdai; Jin, Xin; Dou, Haoxuan; Hu, Chengming; Li, Baorun; Liu, Zhanwang; Xu, Ruiyan; Fang, Jianjie; Zhang, Xin; Yang, Zhenjie; Yang, Xue; Gao, Chen; Yan, Junchi; Li, Yong; Wu, Wei",
    "affiliations": "1Manifold AI 2Tsinghua University 3Shanghai Jiao Tong University",
    "contribution": "In this paper, we introduce WorldScape Policy 2.0, a controllable WAM with reasoning-augmented long short-term memory.",
    "abstract": "World Action Models (WAMs) offer a promising paradigm for robotic manipulation by jointly modeling visual state transitions and robot actions. However, existing WAMs are constrained by limited temporal context, coarse episode-level language supervision, and predominantly text-only conditioning, which hinder task-progress tracking and fine-grained language-video-action grounding while limiting visual-context reasoning and cross-embodiment transfer. In this paper, we introduce WorldScape Policy 2.0, a controllable WAM with reasoning-augmented long short-term memory. Its causal short-term visual memory supplies recent observations as DiT prefill to preserve local interaction dynamics, while its long short-term event memory organizes historical VLM outputs into global-history, local-active, and event-boundary representations for progress-aware retrieval. The retrieved history augments perception and autoregressively generated planning tokens, yielding an implicit subgoal condition for autonomous planning; semantic forcing further transfers event-level instruction semantics into this latent planning pathway. To establish fine-grained multimodal controllability, we construct ManipEvent-5M, an event-grounded embodied pretraining dataset containing nearly 5 million event segments with aligned action trajectories, episode-level task instructions, segment-level subtask captions, goal images, and video demonstrations. These designs provide a unified interface for autonomous planning from high-level instructions and controllable execution from fine-grained text, goal-image, or video-context prompts. Experiments in both simulation and real-world platforms demonstrate superior capabilities in long-horizon autonomous planning, fine-grained instruction following and in-context adaptation.",
    "submittedDate": "2026-07-21",
    "primaryCategory": "Memory WAM",
    "secondaryCategories": [
      "Latent / Representation WAM",
      "Multimodal / Tactile WAM"
    ],
    "bibtexKey": "WorldScapePolicy20",
    "arxivUrl": "https://arxiv.org/abs/2607.18840",
    "codeUrls": [
      "https://github.com/manifoldai-research/WorldScape-Policy"
    ],
    "projectUrl": "https://manifoldai-research.github.io/WorldScape-Policy/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2607.18840",
    "pdfUrl": "https://arxiv.org/pdf/2607.18840",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{WorldScapePolicy2,\n  title={WorldScape Policy 2.0: Empowering Steerable World Action Modeling with Reasoning-Augmented Memory},\n  author={Su, Haisheng and Liu, Zongdai and Jin, Xin and Dou, Haoxuan and Hu, Chengming and Li, Baorun and Liu, Zhanwang and Xu, Ruiyan and Fang, Jianjie and Zhang, Xin and Yang, Zhenjie and Yang, Xue and Gao, Chen and Yan, Junchi and Li, Yong and Wu, Wei},\n  journal={arXiv:2607.18840},\n  year={2026}\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "联合视频动作建模",
      "记忆与长时序",
      "泛化与动作对齐"
    ],
    "architecture": "One Model",
    "predictionParadigm": "联合预测",
    "quadrant": "Q1 · One Model × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2607.17454",
    "title": "Test-Time Scaling for World Action Models via Zero-Shot Geometric Evaluation",
    "authors": "Zhao, Zesen; Cho, Minkyoung; shen, Hui; Zheng, Boyuan; Gao, Kunxiao; Cao, Yulong; Mao, Z. Morley",
    "affiliations": "University of Michigan NVIDIA;",
    "contribution": "We propose \\methodgated, a training-free selective test-time scaling framework for WAMs.",
    "abstract": "Test-time scaling improves foundation-model inference by spending additional computation, but robot control requires deciding whether extra compute is useful before executing an action. World Action Models (WAMs) make this decision natural: each rollout exposes both an action chunk and predicted future observations. We propose \\methodgated, a training-free selective test-time scaling framework for WAMs. We first instantiate \\method, a fixed-budget Best-of-NN selector that ranks sampled rollouts by cross-view depth reprojection consistency of their predicted futures, computed with a frozen geometry foundation model. \\methodgated\\ adds a lightweight action--future consistency gate that invokes \\method\\ only when the initial rollout appears internally inconsistent. Across five benchmark--backbone settings on RoboCasa, LIBERO Long, and RoboTwin~2.0, fixed-budget \\method\\ improves N=8N{=}8 task success in every setting, e.g., raising the RoboCasa group average from 66.3%66.3\\% to 68.4%68.4\\% with Cosmos Policy and from 80.8%80.8\\% to 82.5%82.5\\% with X-WAM. With gating enabled, \\methodgated\\ recovers on average 74.8%74.8\\% of the always-on success gain while triggering additional sampling on only 26.2%26.2\\% of decision points. Offline diagnostics show that cross-view reprojection is a strong task-label-free selector, and we identify false low-score selections as a failure mode that helps explain why performance can saturate or degrade as NN increases.",
    "submittedDate": "2026-07-19",
    "primaryCategory": "3D/4D WAM",
    "secondaryCategories": [
      "Evaluation / Survey / Theory",
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "TTSWAM",
    "arxivUrl": "https://arxiv.org/abs/2607.17454",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2607.17454",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "三维多视角建模",
      "高效推理与实时控制",
      "策略后训练与WM-RL"
    ],
    "architecture": "不适用",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2607.13960",
    "title": "GigaWorld-Policy-0.5: A Faster and Stronger WAM Empowered by AutoResearch",
    "authors": "GigaWorld Team; Angen Ye; Angyuan Ma; Boyuan Wang; Chaojun Ni; Fangzheng Ye; Guan Huang; Guo Li; Guosheng Zhao; Haodong Yan; Hengtao Li; Jiwen Lu; Kai Wang; Mingming Yu; Qitang Hu; Qiuping Deng; Songling Liu; Xiaoyu Tian; Xiaofeng Wang; Xinyu Zhou; Xiuwei Xu; Xinze Chen; Yang Wang; Yejun Zeng; Yifan Chang; Yun Ye; Zhenyu Wu; Zhanqian Wu; Zheng Zhu",
    "affiliations": "Tsinghua University",
    "contribution": "Building upon this framework, we present GigaWorld-Policy-0.5, an enhanced action-centered WAM designed for more efficient robot control.",
    "abstract": "World Action Models (WAMs) improve robot policy learning by jointly modeling actions and future visual observations, using future scene evolution as dense supervision for physically grounded action generation. However, a common design in existing WAMs is to explicitly generate future videos at inference time, incurring substantial computational overhead and hindering real-time closed-loop deployment. GigaWorld-Policy addresses this issue with an action-centered formulation, where future visual dynamics are used during training while action-only decoding is used at inference time. Building upon this framework, we present GigaWorld-Policy-0.5, an enhanced action-centered WAM designed for more efficient robot control. During pretraining, GigaWorld-Policy-0.5 adopts a mixed Action-Conditioned World Modeling (AC-WM) and WAM training strategy. This strengthens the coupling between visual dynamics and robot actions and improves the transferability of action representations for downstream policy learning. For efficient inference, GigaWorld-Policy-0.5 introduces a Mixture-of-Transformers architecture that separates visual dynamics modeling and action generation into specialized experts, reducing active computation during action-only inference and achieving 85 ms inference latency on a local RTX 4090 setup. In addition, we employ an agent-based AutoResearch pipeline to systematically search training configurations, enabling more efficient identification of optimal experimental setups while reducing the time and manual intervention required for hyperparameter tuning. Experiments and ablations show that GigaWorld-Policy-0.5 preserves the training benefits of future visual dynamics while improving inference efficiency for robot control.",
    "submittedDate": "2026-07-17",
    "primaryCategory": "Real-Time / Efficient WAM",
    "secondaryCategories": [],
    "bibtexKey": "GigaWorldPolicy05",
    "arxivUrl": "https://arxiv.org/abs/2607.13960",
    "codeUrls": [
      "https://github.com/open-gigaai/giga-world-policy"
    ],
    "projectUrl": "https://open-gigaai.github.io/giga-world-policy/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2607.13960",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "高效推理与实时控制",
      "泛化与动作对齐"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2607.15207",
    "title": "BadWAM: When World-Action Models Dream Right but Act Wrong",
    "authors": "Li, Qi; Yang, Xingyi; Wang, Xinchao",
    "affiliations": "1 National University of Singapore; 2 The Hong Kong Polytechnic University",
    "contribution": "We introduce BadWAM, a unified framework for modeling and evaluating World-Action Drift Attacks: a new class of WAM-specific adversarial attacks that use small visual perturbations to break the alignment between what a WAM imagines and what it executes.",
    "abstract": "World-action models (WAMs) are emerging as a promising foundation for embodied control: rather than predicting actions alone, they learn representations that couple action generation with future world prediction. This coupling is often viewed as a source of robustness, interpretability, and safety, as a robot's action can in principle be checked against its imagined future. In this paper, we show that this assumption is fragile. We introduce BadWAM, a unified framework for modeling and evaluating World-Action Drift Attacks: a new class of WAM-specific adversarial attacks that use small visual perturbations to break the alignment between what a WAM imagines and what it executes. BadWAM characterizes this attack surface along two natural criteria: attack strength and stealthiness. When the adversary prioritizes disruption, BadWAM instantiates an action-only adversarial attack, which directly drives the model toward task-failing actions. When the adversary additionally prioritizes stealth, BadWAM instantiates an imagination-preserving adversarial attack, which seeks to induce harmful action shifts while keeping the model's predicted future close to its clean imagination. Together, these two attacks capture a spectrum of WAM-specific failures: from overt action hijacking to stealthier cases where the model appears to imagine a plausible future but executes a desynchronized action. We evaluate BadWAM across different variants of WAMs. Results show that our attacks substantially reduce task success rates under closed-loop execution. For example, our action-only attack reduces the model performance from 96.5% to 43.1% success. The results of our imagination-preserving attack further exposes a WAM-specific vulnerability: moderate future-preserving regularization can maintain strong attack performance while reducing future imagination drift.",
    "submittedDate": "2026-07-16",
    "primaryCategory": "General WAM",
    "secondaryCategories": [],
    "bibtexKey": "BadWAM",
    "arxivUrl": "https://arxiv.org/abs/2607.15207",
    "codeUrls": [
      "https://github.com/LiQiiiii/BadWAM"
    ],
    "projectUrl": "https://liqiiiii.github.io/BadWAM",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2607.15207",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "Related resources",
    "subcategories": [
      "评测协议与诊断"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2607.14997",
    "title": "AeroAct: Action-Centered World-Action Models for Language-Conditioned Quadrotor Flight",
    "authors": "Zhang, Xinhong; Zhu, Qiyuan; Huang, Yubo; Chen, Haolin; Wang, Runqing; Mo, Yuhao; Chen, Zhongxin; Hu, Yu; Wang, Xinjiang; Sun, Jian; Wang, Gang",
    "affiliations": "1 School of Automation, Beijing Institute of Technology; 2 School of Mechanical Engineering, Beijing Institute of Technology",
    "contribution": "We present AeroAct, an action-centered world-action model (WAM) for quadrotor navigation. To obtain aligned visual, state, language, and dynamically feasible action data, we build a DiffAero-based pipeline with complementary Isaac Lab and 3D Gaussian splatting renderers.",
    "abstract": "Language-conditioned quadrotor flight requires a policy to ground semantic goals, anticipate the visual consequences of ego-motion, and output control references that remain smooth and dynamically executable under rapidly changing first-person views. Existing aerial vision-language navigation and vision-language-action methods commonly use discrete actions, high-level waypoints, or instantaneous velocity commands, which provide limited supervision about how flight actions change future observations. We present AeroAct, an action-centered world-action model (WAM) for quadrotor navigation. To the best of our knowledge, AeroAct is the first WAM instantiated and demonstrated for real-world aerial flight. The model adapts a pretrained video diffusion Transformer to predict local trajectory-action chunks from egocentric visual history, proprioception, and language. Future first-person frames are used during training as dense consequence supervision, while deployment directly decodes actions without generating future video. To obtain aligned visual, state, language, and dynamically feasible action data, we build a DiffAero-based pipeline with complementary Isaac Lab and 3D Gaussian splatting renderers. We further introduce a low-cost handheld collection device that couples camera observations with motion estimates to recreate flight-like egocentric trajectories, and a self-guidance procedure that improves temporal consistency across overlapping trajectory chunks. Closed-loop simulation and real-world experiments show that temporal visual context improves target tracking and object-search performance, and that WAM-based policies can be executed on a physical quadrotor.",
    "submittedDate": "2026-07-16",
    "primaryCategory": "3D/4D WAM",
    "secondaryCategories": [
      "Multimodal / Tactile WAM",
      "Navigation / Driving / Domain WAM"
    ],
    "bibtexKey": "AeroAct",
    "arxivUrl": "https://arxiv.org/abs/2607.14997",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2607.14997",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "导航",
      "联合视频动作建模",
      "高效推理与实时控制"
    ],
    "architecture": "One Model",
    "predictionParadigm": "联合预测",
    "quadrant": "Q1 · One Model × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2607.14943",
    "title": "Steering Robustness into World Action Models via Mechanistic Interpretability and Optimal Control",
    "authors": "Hong, Jihoon; Skifstad, Julian; Dai, Qiyue; Chan, Alice; Chou, Glen",
    "affiliations": "Georgia Institute of Technology",
    "contribution": "World Action Models (WAMs) enable semantically- and physically-informed control but are brittle under distribution shift. In this work, we use mechanistic interpretability to study how robustness-relevant perturbations are represented in WAM activation space.",
    "abstract": "World Action Models (WAMs) enable semantically- and physically-informed control but are brittle under distribution shift. In this work, we use mechanistic interpretability to study how robustness-relevant perturbations are represented in WAM activation space. Comparing activations across successful and unsuccessful rollouts, we find some WAM architectures exhibit low-dimensional linear separability for robustness-critical features, while others do not. This motivates the use of contrastive activation directions for training-free WAM steering. We also show that local linearity in WAM activation dynamics enables efficient feedback steering via model-based optimal control, yielding World-Action Linear Quadratic Regulator (WA-LQR), a minimally-invasive reduced-order LQR controller. Via mechanistic evaluations, we predict strong steerability in the Cosmos-Policy and DiT4DiT models but weak steerability in LingBot-VA, consistent with steering intervention results. On Cosmos-Policy and DiT4DiT, WA-LQR generalizes contrastive directions to new tasks and improves robustness to camera, gripper, and visual-noise perturbations over unsteered and prompt steering baselines.",
    "submittedDate": "2026-07-16",
    "primaryCategory": "Real-Time / Efficient WAM",
    "secondaryCategories": [],
    "bibtexKey": "WALQR",
    "arxivUrl": "https://arxiv.org/abs/2607.14943",
    "codeUrls": [
      "https://github.com/trustworthyrobotics/steering_robustness_WAMs"
    ],
    "projectUrl": "https://trustworthyrobotics.github.io/steering_robust_wam_site/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2607.14943",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "泛化与动作对齐",
      "高效推理与实时控制"
    ],
    "architecture": "不适用",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2607.13017",
    "title": "FlowWAM: Optical Flow as a Unified Action Representation for World Action Models",
    "authors": "Chen, Yixiang; Li, Peiyan; Xu, Yuan; Ma, Qisen; Yang, Jiabing; Wang, Kai; Yang, Jianhua; An, Dong; Guan, He; Liu, Gaoteng; Si, Jianlou; Huang, Jun; Liu, Jing; Liu, Nianfeng; Huang, Yan; Wang, Liang",
    "affiliations": "New Laboratory of Pattern Recognition (NLPR); Institute of Automation, Chinese Academy of Sciences; School of Artificial Intelligence, University of Chinese Academy of Sciences; Alibaba Group",
    "contribution": "On WorldArena world modeling, it achieves the best overall EWMScore (63.71) with an 18.4% relative improvement in trajectory accuracy.",
    "abstract": "World Action Models (WAMs) are able to leverage pretrained video generators for both world modeling and action prediction. However, directly leveraging such video generators for control raises a new challenge: how to represent actions in a suitable form that aligns with pretrained video generators while carrying enough motion cues for accurate control. Existing numerical actions fail to satisfy the former, and prior visual action representations overlook the temporal motion structure across frames. We address this issue with FlowWAM, a dual-stream diffusion framework that adopts optical flow as a unified, video-native action representation. Flow videos share the same format as RGB videos and encode rich per-pixel displacement. By jointly modeling them within a shared pretrained video generator, FlowWAM can naturally implement two modes of WAMs. In policy mode, FlowWAM generates flow for action prediction, while in world-model mode, it uses target flow sequences to guide future video generation. Moreover, since flow can be easily extracted from raw videos without action labels, FlowWAM can leverage large-scale action-unlabeled video datasets for pretraining. We empirically find that our flow-based action representation delivers gains across both modes. On RoboTwin manipulation, FlowWAM raises the success rate to 92.94% on the Clean setting and 92.14% on Random, outperforming both VLA and WAM baselines. On WorldArena world modeling, it achieves the best overall EWMScore (63.71) with an 18.4% relative improvement in trajectory accuracy. More results can be found on our project website: https://flow-wam.github.io .",
    "submittedDate": "2026-07-14",
    "primaryCategory": "General WAM",
    "secondaryCategories": [],
    "bibtexKey": "FlowWAM",
    "arxivUrl": "https://arxiv.org/abs/2607.13017",
    "codeUrls": [
      "https://github.com/YixiangChen515/FlowWAM"
    ],
    "projectUrl": "https://flow-wam.github.io/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2607.13017",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "视觉规划与IDM",
      "泛化与动作对齐"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2607.11689",
    "title": "From World Action Models to Embodied Brains: A Roadmap for Open-World Physical Intelligence",
    "authors": "Liang, Yuanzhi; Zhan, Xufeng; Huang, Haibin; Zhang, Chi; Li, Xuelong",
    "affiliations": "a co-evolution roadmap for scalable physical intelligence. At its center is theembodied brain, a",
    "contribution": "Building on this analysis, we propose a co-evolution roadmap for physical intelligence centered on the \\emph{embodied brain}, a long-term model target for integrating multimodal context, comparing candidate interventions, and issuing state-transition or capability requests rather than direct actuator commands.",
    "abstract": "Artificial general intelligence ultimately requires agents that can reason and act in the physical world. Action models, vision-language-action policies, and world models have advanced this goal, while World Action Models (WAMs) are particularly promising because they connect candidate interventions with predicted consequences. However, progress remains fragmented: models use incompatible action spaces and prediction targets, datasets and tasks follow different conventions, and runtime systems expose limited interfaces for reuse and evaluation. We review the evolution toward WAMs and organize these limitations into three coupled gaps: model roles and representations, objectives and standardization, and system composition. Building on this analysis, we propose a co-evolution roadmap for physical intelligence centered on the \\emph{embodied brain}, a long-term model target for integrating multimodal context, comparing candidate interventions, and issuing state-transition or capability requests rather than direct actuator commands. WAMs provide promising prototypes for its predictive functions, while a physical harness grounds model outputs through tools, controllers, verification, and trace logging. Shared contracts align heterogeneous models, data, tasks, and embodiments, and closed-loop post-training converts verified interaction into reusable experience. Together, these components define a modular physical-intelligence stack for adaptive and self-improving embodied agents.",
    "submittedDate": "2026-07-13",
    "primaryCategory": "Evaluation / Survey / Theory",
    "secondaryCategories": [
      "Multimodal / Tactile WAM"
    ],
    "bibtexKey": "coevolution",
    "arxivUrl": "https://arxiv.org/abs/2607.11689",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2607.11689",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "Related resources",
    "subcategories": [
      "综述与技术资源"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2607.11270",
    "title": "Towards Predictive, Aligned, and Scalable Robot Learning",
    "authors": "Tang, Peijun; Xie, Shangjin; Huang, Baifu; Sun, Binyan; Yang, Haotian; Luo, Kuncheng; Jin, Weiqi; Fang, Shilin; Wang, Jianan",
    "affiliations": "Not identified",
    "contribution": "We introduce Lumo-2, a latent world-action model that generates actions by reasoning over world dynamics in latent space. Central to our approach is the hypothesis that action generation quality is governed by the geometry of the latent space.",
    "abstract": "Learning, at its core, extends beyond memorization to the ability to reason and solve novel problems by navigating a space of possibilities. We introduce Lumo-2, a latent world-action model that generates actions by reasoning over world dynamics in latent space. The learned latent world dynamics capture physically grounded visual transitions, naturally encoding future possibilities and providing a unified substrate for cross-modal alignment. This formulation enables predictive reasoning akin to world modelling while remaining lightweight and focused on physical dynamics relevant to control. Central to our approach is the hypothesis that action generation quality is governed by the geometry of the latent space. We observe that standard reconstruction-based action tokenization objectives induce representations biased toward low-level signal fidelity, leading to misalignment between reconstruction quality and downstream control performance. To address this limitation, we propose a multi-stage modality pre-alignment strategy in which action representations are progressively aligned with latent world dynamics, vision, and language. This process enforces cross-modal consistency, promotes abstraction, and induces a structured latent space for predictive reasoning. We provide a systematic empirical study of latent world modelling and modality alignment, analyzing their roles in scaling laws and out-of-distribution generalization. Results show that Lumo-2 consistently outperforms strong vision-language-action (VLA) and world-action model (WAM) baselines, with gains on challenging real-world tasks requiring temporal reasoning, physical understanding, or high control complexity, including long-horizon and dexterous manipulation. These findings suggest that structured multimodal alignment and predictive reasoning are fundamental principles for advancing embodied intelligence.",
    "submittedDate": "2026-07-13",
    "primaryCategory": "3D/4D WAM",
    "secondaryCategories": [
      "Evaluation / Survey / Theory",
      "Latent / Representation WAM",
      "Multimodal / Tactile WAM",
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "Lumo2",
    "arxivUrl": "https://arxiv.org/abs/2607.11270",
    "codeUrls": [],
    "projectUrl": "https://www.astribot.com/research/Lumo2",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2607.11270",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "潜空间预测与JEPA",
      "视觉规划与IDM",
      "泛化与动作对齐"
    ],
    "architecture": "One Model",
    "predictionParadigm": "IDM",
    "quadrant": "Q2 · One Model × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2607.10655",
    "title": "Artificial Foveated Perception for Mitigating Shortcut Learning in Robotic Foundation Models",
    "authors": "Sun, Xiatao; Zhuang, Yuan; Negrete, Mateo Sanchez Lopez; Coldea, Matei-Victor; Liang, Chen; Zhang, Haoyang; Liu, Che; Zeng, Ziyao; Li, Shawn; Wang, Qian; Miao, Fei; Rakita, Daniel",
    "affiliations": "1 Yale University 2 University of Connecticut; 3 Peking University 4 Imperial College London 5 Digients",
    "contribution": "We propose Artificial Foveated Perception (AFP), a lightweight, policy-agnostic module that takes the same vision and language inputs as Vision-Language-Action and World Action Model pipelines and predicts task-conditioned masks over relevant objects, the robot, and other action-critical regions. We evaluate AFP across state-of-the-art robotic foundation models and show that foveated perception reduces fine-tuning time, suppresses overfitting, and improves generalization under environmental perturbations.",
    "abstract": "Robotic foundation models have recently made substantial progress in multi-task capability, cross-embodiment transfer, and language-conditioned control. Yet robust deployment across diverse real-world settings remains difficult, in part because policies often fail to distinguish causally relevant visual structure from spurious scene-level correlations. We identify this failure mode as shortcut learning: the tendency to exploit predictive but non-causal correlations in the training distribution rather than the task-relevant visual evidence that determines successful action. Although shortcut learning has been extensively studied in computer vision and broader machine learning, its role in robotic foundation models remains comparatively underexplored. We propose Artificial Foveated Perception (AFP), a lightweight, policy-agnostic module that takes the same vision and language inputs as Vision-Language-Action and World Action Model pipelines and predicts task-conditioned masks over relevant objects, the robot, and other action-critical regions. We use these masks primarily as an auxiliary grounding signal during fine-tuning, aligning policy attention with task-relevant regions while leaving the core architecture unchanged. After fine-tuning, the policy executes on the original observation stream without requiring AFP in the control loop. We evaluate AFP across state-of-the-art robotic foundation models and show that foveated perception reduces fine-tuning time, suppresses overfitting, and improves generalization under environmental perturbations. Ablations over mask quality and grounding-loss design further show that these gains arise from directing policy learning toward task-relevant visual evidence. These results suggest that task-conditioned foveated perception is a practical mechanism for making robotic foundation models more robust, data-efficient, and scalable.",
    "submittedDate": "2026-07-12",
    "primaryCategory": "Real-Time / Efficient WAM",
    "secondaryCategories": [],
    "bibtexKey": "AFP",
    "arxivUrl": "https://arxiv.org/abs/2607.10655",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2607.10655",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "VLA",
    "subcategories": [
      "训练优化与蒸馏",
      "动作策略基础"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2606.27677",
    "title": "DIM-WAM: World-Action Modeling with Diverse Historical Event Memory",
    "authors": "Wang, Kai; Gu, Zhaopeng; Chen, Yixiang; Xu, Yuan; Ma, Qisen; Yang, Jiabing; Li, Zhaowen; Huang, Yan; Wang, Liang; Su, Peng",
    "affiliations": "1 Institute of Automation, Chinese Academy of Sciences (CASIA), Beijing, China. 2 Shenzhen Yinwang Intelligent Technology Co., Ltd., Shenzhen, China. 3 FiveAges, Beijing, China",
    "contribution": "To address long-term forgetting and poor awareness of the global task state, we introduce DiM-WAM, a memory-augmented world-action model that integrates multi-scale historical context, local future dynamics, and global task progress.",
    "abstract": "World-action models have shown promising robot-manipulation performance by jointly predicting future visual states and actions. However, existing methods mainly rely on short-term history and short-horizon future prediction, which is insufficient for long-horizon tasks whose correct execution depends on earlier observations and task progress. Such temporally dependent tasks require effective use of complementary temporal information, including recent local context, cross-stage historical events, immediate future dynamics, and global task progress. To address long-term forgetting and poor awareness of the global task state, we introduce DiM-WAM, a memory-augmented world-action model that integrates multi-scale historical context, local future dynamics, and global task progress. The memory extracts compact visual event information from real observations, updates multiple memory banks through independent similarity-based merging, and then reads the bank-identity- and time-embedded long-term context to condition video and action denoising. A progress-supervision objective further encourages memory tokens to encode not only completed historical events but also the current task stage and its implications for the remaining task. On RMBench, DiM-WAM raises average success from 28.4% with LingBot-VA to 69.8%, exceeding the explicit-memory Mem-0 baseline at 42.0%. On four real-world Franka tasks, it improves average stage success from 70.7% to 91.5% and full-task success from 52.5% to 80.0%. Project page: https://wangkai-casia.github.io/dim-wam.",
    "submittedDate": "2026-07-12",
    "primaryCategory": "Memory WAM",
    "secondaryCategories": [
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "DIMWAM",
    "arxivUrl": "https://arxiv.org/abs/2606.27677",
    "codeUrls": [],
    "projectUrl": "https://wangkai-casia.github.io/dim-wam/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2606.27677",
    "pdfUrl": "https://arxiv.org/pdf/2606.27677",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{DIMWAM,\n  title={DIM-WAM: World-Action Modeling with Diverse Historical Event Memory},\n  author={Wang, Kai and Gu, Zhaopeng and Chen, Yixiang and Xu, Yuan and Ma, Qisen and Yang, Jiabing and Li, Zhaowen and Huang, Yan and Wang, Liang and Su, Peng},\n  journal={arXiv:2606.27677},\n  year={2026}\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "记忆与长时序",
      "联合视频动作建模"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2607.08877",
    "title": "FlowDAgger: Human-in-the-Loop Adaptation of Generative Robot Policies in Latent Space",
    "authors": "Murray, Michael; Chen, Daphne; Bagaria, Simran; Fortier, Dean; Hellebrekers, Tess; Mullins, Galen; Gajarla, Harshavardhan; Mees, Oier; Cakmak, Maya; Kolobov, Andrey",
    "affiliations": "Not identified",
    "contribution": "We present FlowDAgger, a sample- and compute-efficient method for adapting frozen generative robot policies from human interventions in latent space. FlowDAgger outperforms supervised fine-tuning and latent-space RL baselines and preserves pretrained skills on held-out tasks, offering a practical path for adapting robot foundation models in the real world.",
    "abstract": "Pretrained generative robot policies based on flow matching and diffusion have achieved impressive results across a wide range of manipulation tasks. Yet real-world deployments routinely expose failure modes outside the pretraining distribution. Closing these gaps typically requires large-scale data collection or online reinforcement learning on physical hardware, which is impractical for rapid and safe adaptation. We present FlowDAgger, a sample- and compute-efficient method for adapting frozen generative robot policies from human interventions in latent space. Our key idea is action inversion: each human expert action is mapped to the noise that would have produced it under the frozen base policy, using reverse-time integration followed by local refinement. The resulting inverted noise provides supervision for a lightweight latent policy that steers the base model at deployment time, enabling rapid skill acquisition while preserving its behavioral priors. We evaluate FlowDAgger in simulation and on real-world bimanual and single-arm manipulation, adapting both action-head VLAs and world-action models from a handful of interventions. FlowDAgger outperforms supervised fine-tuning and latent-space RL baselines and preserves pretrained skills on held-out tasks, offering a practical path for adapting robot foundation models in the real world. Website: https://microsoft.github.io/FlowDAgger",
    "submittedDate": "2026-07-09",
    "primaryCategory": "Latent / Representation WAM",
    "secondaryCategories": [
      "Real-Time / Efficient WAM",
      "WAM + RL"
    ],
    "bibtexKey": "FlowDAgger",
    "arxivUrl": "https://arxiv.org/abs/2607.08877",
    "codeUrls": [
      "https://github.com/microsoft/FlowDAgger"
    ],
    "projectUrl": "https://microsoft.github.io/FlowDAgger",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2607.08877",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "训练优化与蒸馏",
      "动作策略基础"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2607.08448",
    "title": "Harness VLA: Steering Frozen VLAs into Reliable Manipulation Primitives via Memory-Guided Agents",
    "authors": "Yixian Zhang; Huanming Zhang; Feng Gao; Xiao Li; Zhihao Liu; Chunyang Zhu; Jiaxing Qiu; Yuchen Yan; Jiyuan Liu; Wenhao Tang; Zhengru Fang; Yi Nie; Changxu Wei; Yu Wang; Wenbo Ding; Chao Yu",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2026-07-09",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "zhang2026harnessvla",
    "arxivUrl": "https://arxiv.org/abs/2607.08448",
    "codeUrls": [
      "https://github.com/RLinf/RPent"
    ],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2607.08448",
    "pdfUrl": "https://arxiv.org/pdf/2607.08448",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{zhang2026harnessvla,\n  title={Harness VLA: Steering Frozen VLAs into Reliable Manipulation Primitives via Memory-Guided Agents},\n  author={Zhang, Yixian and Zhang, Huanming and Gao, Feng and Li, Xiao and Liu, Zhihao and Zhu, Chunyang and Qiu, Jiaxing and Yan, Yuchen and Liu, Jiyuan and Tang, Wenhao and Fang, Zhengru and Nie, Yi and Wei, Changxu and Wang, Yu and Ding, Wenbo and Yu, Chao},\n  journal={arXiv:2607.08448},\n  year={2026}\n}",
    "majorCategory": "VLA",
    "subcategories": [
      "分层与双系统VLA"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2607.08127",
    "title": "Understanding and Mitigating the Video-Action Generalization Gap via Temporal Ratio",
    "authors": "Mishra, Utkarsh A.; Chen, Yongxin; Xu, Danfei; Liu, Yang; Chen, Xi; Mao, Jiayuan",
    "affiliations": "Amazon FAR; Work done during an internship at Amazon FAR",
    "contribution": "To explain this behavior, we introduce the Temporal Ratio (TR), an attention-based measure of how strongly the action head relies on future latent rollouts relative to the anchored current frame. Finally, based on these findings, we propose an inference-time adaptive guidance method, which exploits this intrinsic feature attention pattern to dynamically amplify compositional video conditioning signals precisely when the policy relies on future rollouts.",
    "abstract": "Generative video foundation models exhibit strong compositional priors, yet world-action models (WAMs) and video-action models (VAMs) often lose these priors after finetuning on robotic action data. We refer to this discrepancy as the video-action generalization gap. In this paper, we systematically investigate this gap by evaluating a comprehensive design space of VAMs, demonstrating that standard design choices yield no emergent explanation pattern. To explain this behavior, we introduce the Temporal Ratio (TR), an attention-based measure of how strongly the action head relies on future latent rollouts relative to the anchored current frame. TR has two key properties: first, a model's structural reliance on future-predictive latents, measured via TR, acts as a predictor of its compositional generalization capacity; second, it natively fluctuates based on task phase, shifting attention to future frames during planning and reverting to the present frame for precise manipulation. Finally, based on these findings, we propose an inference-time adaptive guidance method, which exploits this intrinsic feature attention pattern to dynamically amplify compositional video conditioning signals precisely when the policy relies on future rollouts. Evaluated on the LIBERO benchmark and real-world tasks, our approach mitigates the OOD-ID compositional generalization gap. More details: https://umishra.me/temporal-ratio/",
    "submittedDate": "2026-07-09",
    "primaryCategory": "Latent / Representation WAM",
    "secondaryCategories": [
      "Evaluation / Survey / Theory"
    ],
    "bibtexKey": "VAMs",
    "arxivUrl": "https://arxiv.org/abs/2607.08127",
    "codeUrls": [],
    "projectUrl": "https://umishra.me/temporal-ratio/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2607.08127",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "视觉规划与IDM",
      "泛化与动作对齐"
    ],
    "architecture": "不适用",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2607.06988",
    "title": "WAM-TTT: Steering World-Action Models by Watching Human Play at Test Time",
    "authors": "Feng, Yusen; Han, Bingchen; Lyu, Jiangran; Liu, Kai; Zheng, Yixin; Wan, Yuxuan; Liu, Weiheng; Han, Sun; Li, Ruiqin; Zhang, Yulong; Liu, Fangfu; Shi, Xuesong; Liu, Libin; Wang, Yizhou; Zhang, Zhizheng; Wang, He",
    "affiliations": "1 Peking University; 4 Tsinghua University",
    "contribution": "We present WAM-TTT, a test-time training framework for steering world action models from raw human videos. To make this memory useful for control, we introduce a meta-training stage that aligns human demonstrations with robot behaviors using paired human-robot data and a key--value memory reconstruction objective.",
    "abstract": "Steering robot foundation models (RFMs) toward new task variants or user-preferred behaviors remains challenging, often requiring additional robot demonstrations, task-specific fine-tuning, or long-context conditioning. We present WAM-TTT, a test-time training framework for steering world action models from raw human videos. Rather than treating human videos as trajectories to imitate, WAM-TTT absorbs them into a lightweight adaptive memory inside a frozen WAM through self-supervised video prediction. To make this memory useful for control, we introduce a meta-training stage that aligns human demonstrations with robot behaviors using paired human-robot data and a key--value memory reconstruction objective. At test time, only unlabeled human videos are required to adapt the memory, while the pretrained WAM remains frozen. This enables efficient and reusable steering without robot actions, human-side annotations, or task-specific fine-tuning, while preserving the generalization ability of the foundation model. Extensive experiments show that WAM-TTT consistently outperforms in-context human-video conditioning baselines across diverse manipulation tasks and generalization settings.",
    "submittedDate": "2026-07-09",
    "primaryCategory": "Memory WAM",
    "secondaryCategories": [
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "WAMTTT",
    "arxivUrl": "https://arxiv.org/abs/2607.06988",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2607.06988",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "记忆与长时序",
      "泛化与动作对齐"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "待核实",
    "quadrant": "待核实",
    "classificationStatus": "部分待核实"
  },
  {
    "id": "2607.08436",
    "title": "EgoWAM: World Action Models Beyond Pixels with In-the-Wild Egocentric Human Data",
    "authors": "Li, Baoyu; Yin, Xinchen; Lin, Mengying; Zhang, Yixin; Xu, Danfei",
    "affiliations": "Not identified",
    "contribution": "We introduce EgoWAM, a controlled human-robot co-training framework that fixes the policy backbone, action head, and data mixture while varying only the world prediction target, comparing Pixel, DINO, and 3D motion flow.",
    "abstract": "Egocentric human data offers scalable supervision for robot manipulation. However, behavior cloning entangles transferable content like objects, scenes, and task semantics, with non-transferable factors like human morphology, head motion, and behavioral style. We study whether World Action Models (WAMs) provide a better training signal by requiring policies to predict not only actions, but also how the scene evolves. The central question is what world representation best enables human-to-robot transfer. We hypothesize that an effective world target should abstract appearance, capture agent-invariant physical effects, and separate camera motion from environment change. We introduce EgoWAM, a controlled human-robot co-training framework that fixes the policy backbone, action head, and data mixture while varying only the world prediction target, comparing Pixel, DINO, and 3D motion flow. Across three real-world bimanual tasks, WAM co-training scales more effectively with in-the-wild egocentric human data than behavior cloning. Pixel-based prediction transfers weakly, while DINO and 3D flow yield substantial gains: DINO improves out-of-distribution object and scene generalization by up to 4x, and 3D flow improves in-domain performance by 20-30%. More details: https://gatech-rl2.github.io/egowam.github.io",
    "submittedDate": "2026-07-08",
    "primaryCategory": "3D/4D WAM",
    "secondaryCategories": [],
    "bibtexKey": "EgoWAM",
    "arxivUrl": "https://arxiv.org/abs/2607.08436",
    "codeUrls": [],
    "projectUrl": "https://gatech-rl2.github.io/egowam.github.io",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2607.08436",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "潜空间预测与JEPA",
      "三维多视角建模",
      "泛化与动作对齐"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "联合预测",
    "quadrant": "Q3 · Dual-system × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2606.14048",
    "title": "WAM4D: Fast 4D World Action Model via Spatial Register Tokens",
    "authors": "Li, Ying; Wei, Xiaobao; Cao, Jiajun; Wang, Hao; Chi, Xiaowei; Bai, Chengyu; Sun, Qianpu; Li, Jiajun; Zhang, Xiaojie; Jia, Peidong; Tang, Jian; Han, Sirui; Zhang, Shanghang",
    "affiliations": "1Peking University; 2The Hong Kong University of Science and Technology; 3Beijing Innovation Center of Humanoid Robotics",
    "contribution": "To address the trade-off, we present WAM4D, a fast 4D world action model that uses lightweight spatial register tokens as training-time future-depth readouts to transfer pretrained geometric priors into a causal video-action transformer, then removes the register branch for lightweight action inference. Comprehensive experiments on RoboTwin 2.0 and challenging real-world manipulation tasks show that WAM4D improves spatial consistency and achieves competitive action prediction while maintaining efficient inference.",
    "abstract": "World action models (WAMs) have recently shown promise in jointly modeling future observations and executable robot actions. However, most existing WAMs still operate in 2D video or latent spaces, where visually plausible rollouts miss the 3D spatial constraints and occluded contact geometry required for precise manipulation. While geometric foundation models offer strong priors for recovering dense 3D structure and motion from visual observations, forcing WAMs to predict the dense 4D representation introduces costly geometric decoding and slows down causal action generation. To address the trade-off, we present WAM4D, a fast 4D world action model that uses lightweight spatial register tokens as training-time future-depth readouts to transfer pretrained geometric priors into a causal video-action transformer, then removes the register branch for lightweight action inference. To prevent non-causal shortcuts, we further design causal mixture attention for the Mixture-of-Transformers (MoT) WAM backbone, defining modality-specific visibility among video, action, and geometry tokens. Comprehensive experiments on RoboTwin 2.0 and challenging real-world manipulation tasks show that WAM4D improves spatial consistency and achieves competitive action prediction while maintaining efficient inference.",
    "submittedDate": "2026-07-07",
    "primaryCategory": "3D/4D WAM",
    "secondaryCategories": [
      "Latent / Representation WAM",
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "WAM4D",
    "arxivUrl": "https://arxiv.org/abs/2606.14048",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2606.14048",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "三维多视角建模",
      "高效推理与实时控制"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2607.05468",
    "title": "Learning 4D Geometric Priors for Inference-Efficient World Action Models",
    "authors": "Zhang, Jianjun; Zhu, Jian; Su, Taiyi; Ma, Chong; Huang, Zitai; Xu, Yi; Wang, Hanli",
    "affiliations": "1Tongji University 2AIRC, Midea Group",
    "contribution": "We propose MECo-WAM, a Multi-Expert Co-Training World Action Model that injects action-relevant 4D geometric priors into video-action representations while preserving the original lightweight inference graph. To transfer geometric knowledge into the deployed video-action pathway, we introduce decayed 4D read-mask attention, which provides restricted current-frame geometric guidance early in training and progressively removes this dependency.",
    "abstract": "World Action Models (WAMs) have shown strong potential for robotic manipulation by jointly modeling visual future dynamics and executable action sequences. However, existing video-action co-training methods primarily optimize appearance-oriented video latents, which may insufficiently capture the temporally evolving geometry required for precise manipulation. We propose MECo-WAM, a Multi-Expert Co-Training World Action Model that injects action-relevant 4D geometric priors into video-action representations while preserving the original lightweight inference graph. During training, MECo-WAM combines video and action experts with a lightweight 4D expert supervised by relational targets from a frozen VGGT encoder. Asymmetric expert visibility prevents non-causal shortcuts from auxiliary geometry to action generation. To transfer geometric knowledge into the deployed video-action pathway, we introduce decayed 4D read-mask attention, which provides restricted current-frame geometric guidance early in training and progressively removes this dependency. We further propose action-aware temporal geometric distillation, which aligns within-frame geometric relations and their temporal evolution while emphasizing visual regions most relevant to robot actions. At deployment, all auxiliary 4D components are removed. Experiments on LIBERO (98.2%), RoboTwin 2.0 (92.6%), and challenging real-world manipulation tasks show that MECo-WAM improves manipulation performance without increasing inference cost.",
    "submittedDate": "2026-07-06",
    "primaryCategory": "3D/4D WAM",
    "secondaryCategories": [
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "MECoWAM",
    "arxivUrl": "https://arxiv.org/abs/2607.05468",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2607.05468",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "三维多视角建模",
      "高效推理与实时控制",
      "泛化与动作对齐"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2607.05133",
    "title": "UNIVERSE: Unified Video Action Models for Autonomous Driving with Flexible Mask-Modulated Modality Generation",
    "authors": "Liu, Mengmeng; Zhang, Diankun; Liu, Jiuming; Cui, Jianfeng; Xie, Hongwei; Chen, Guang; Ye, Hangjun; Nex, Francesco; Cheng, Hao; Yang, Michael Ying",
    "affiliations": "University of Twente; University of Cambridge; University of Bath",
    "contribution": "We propose UNIVERSE, a unified video-action model built upon a single mask-modulated Diffusion Transformer. To ensure causal validity and efficient deployment, we introduce a Modality-Decoupling Visibility Mask, which shares historical context across modalities while blocking mutual attention between future video and trajectory tokens.",
    "abstract": "World Action Models (WAMs) have shown strong potential for improving action generalization in autonomous driving by using future video prediction as dense supervision for scene dynamics and temporal causality. However, it remains unclear which architecture better transfers video-modeling benefits to trajectory generation. Existing cascaded or dual-DiT designs separate video imagination from action prediction, weakening the transfer of video-learned world dynamics to the trajectory branch: the action model may still overfit dataset-specific driving priors, while the video model only indirectly regularizes planning. We propose UNIVERSE, a unified video-action model built upon a single mask-modulated Diffusion Transformer. By co-training future video latents and ego-trajectory tokens within shared generative parameters, UNIVERSE allows dense video supervision to directly shape trajectory denoising, leading to stronger cross-domain action generalization. To ensure causal validity and efficient deployment, we introduce a Modality-Decoupling Visibility Mask, which shares historical context across modalities while blocking mutual attention between future video and trajectory tokens. This prevents future-target leakage and enables trajectory-only inference by removing future-video denoising at test time, achieving a 4.3×4.3\\times speedup over joint video-action rollout while maintaining comparable planning accuracy. The same model also supports video-only and joint video-action rollouts. Experiments show that UNIVERSE achieves 91.0 PDMS on NAVSIM (vs. 89.6 for the Two-DiT variant), and demonstrates strong zero-shot transfer to nuScenes and Bench2Drive without fine-tuning, while ablations confirm the importance of single-DiT unification, video co-training, and mask-based modality decoupling.",
    "submittedDate": "2026-07-06",
    "primaryCategory": "Navigation / Driving / Domain WAM",
    "secondaryCategories": [
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "UNIVERSE",
    "arxivUrl": "https://arxiv.org/abs/2607.05133",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2607.05133",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "自动驾驶",
      "高效推理与实时控制"
    ],
    "architecture": "One Model",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2607.04927",
    "title": "DSWAM: A Dual-System World Action Foundation Model for Fine-Grained Robot Manipulation",
    "authors": "Zhu, Jian; Zhang, Jianjun; Su, Taiyi; Liu, Tianbin; Wang, Zhangyuan; Xie, Kai; Huang, Zitai; Ma, Chong; He, Youzhang; Wang, Tianjian; Wang, Hanyang; Ding, Weihao; Xu, Yi",
    "affiliations": "1AIRC, Midea Group, 2Tongji University",
    "contribution": "To address both the decomposition gap and the need for a controlled WAM-VLA comparison, we introduce DSWAM, a Dual-System World Action Foundation Model for fine-grained robot manipulation. To provide a fair real-robot comparison with VLA policies, we build and evaluate DSWAM under the DeMaVLA real-world deformable manipulation setting with matched robot platform, pretraining data, post-training data, and evaluation criteria.",
    "abstract": "World Action Models (WAMs) provide a promising alternative to Vision-Language-Action (VLA) policies by using video-based world modeling as dense supervision for robot action learning. Existing WAMs excel at physically grounded execution, but typically lack the explicit language-level planning interface in VLM-based VLAs for decomposing coarse instructions. Such decomposition becomes important when household tasks involve complex multi-step goals, where coarse user commands need to be converted into sequences of fine-grained executable subtasks. Meanwhile, the field still lacks a fair real-robot comparison between VLA and WAM execution capabilities, since existing systems often differ in data, robot embodiments, and task protocols. To address both the decomposition gap and the need for a controlled WAM-VLA comparison, we introduce DSWAM, a Dual-System World Action Foundation Model for fine-grained robot manipulation. DSWAM keeps a System 1 WAM executor as the default control path and optionally activates a System 2 vision-language subtask planner only when task decomposition is useful. The planner predicts executable subtasks from short-term visual history and a global task prompt, while the WAM executor performs world-aware action generation for each instruction or subtask. The executor is trained with action prediction and video co-training, but inference directly predicts action chunks without explicit future video generation. To make this execution path practical on real robots, we further integrate TensorRT acceleration, asynchronous execution, and real-time chunking (RTC) so that policy queries do not block robot control. To provide a fair real-robot comparison with VLA policies, we build and evaluate DSWAM under the DeMaVLA real-world deformable manipulation setting with matched robot platform, pretraining data, post-training data, and evaluation criteria.",
    "submittedDate": "2026-07-06",
    "primaryCategory": "Real-Time / Efficient WAM",
    "secondaryCategories": [
      "Evaluation / Survey / Theory"
    ],
    "bibtexKey": "DSWAM",
    "arxivUrl": "https://arxiv.org/abs/2607.04927",
    "codeUrls": [],
    "projectUrl": "https://ds-wam.github.io/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2607.04927",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "高效推理与实时控制",
      "记忆与长时序"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2607.00678",
    "title": "ABot-M0.5: Unified Mobility-and-Manipulation World Action Model",
    "authors": "Chen, Ronghan; Yang, Yandan; Tang, Zuojin; Huo, Dongjie; Lin, Tong; Wu, Haoning; Liu, Haoyun; Chen, Yuzhi; Zheng, Lulu; Yuan, Botai; Li, Tianlun; Wang, Mingxin; Qi, Dekang; Hu, Bin; Mei, Wei; Xuan, Yuze; Yang, Haolong; Zhu, Yanqing; Xu, Mu; Ma, Zhiheng; Chang, Xinyuan",
    "affiliations": "AMAP CV Lab",
    "contribution": "We propose ABot-M0.5, a new WAM built on the insight that mobile manipulation requires alignment at three levels: temporal granularity, action space, and train-test consistency. To align temporal granularity, we introduce intermediate latent actions that capture local visual state transitions and serve as an bridging action space between video latents and embodiment-specific controls.",
    "abstract": "Mobile manipulation is a key capability for general-purpose robots, yet remains challenging for current embodied learning methods. VLA policies are typically reactive and lack explicit world modeling, while existing World Action Models (WAMs) are still poorly aligned with the structure of mobile manipulation: they operate on coarse video chunks, model entangled navigation-manipulation actions, and train inverse dynamics under supervision that does not match autoregressive inference. As a result, they often miss fine-grained contact dynamics, suffer from action-distribution conflicts, and accumulate errors over long-horizon rollouts. We propose ABot-M0.5, a new WAM built on the insight that mobile manipulation requires alignment at three levels: temporal granularity, action space, and train-test consistency. To align temporal granularity, we introduce intermediate latent actions that capture local visual state transitions and serve as an bridging action space between video latents and embodiment-specific controls. To align action space, we design a dual-level Mixture-of-Transformers architecture that disentangles both modality representations and heterogeneous action subspaces such as base movement and arm manipulation. To align inference conditions, we propose the dream-forcing training strategy that progressively trains inverse dynamics on model-predicted videos, improving train-test alignment and robustness during autoregressive prediction. Experiments on challenging mobile and fine-grained manipulation benchmarks demonstrate that ABot-M0.5 achieves state-of-the-art performance in both long-horizon task success and finegrained control accuracy. These results highlight the critical importance of granularity-aligned, action-disentangled, and inference-consistent world-action modeling.",
    "submittedDate": "2026-07-06",
    "primaryCategory": "Latent / Representation WAM",
    "secondaryCategories": [
      "Navigation / Driving / Domain WAM"
    ],
    "bibtexKey": "ABotM05",
    "arxivUrl": "https://arxiv.org/abs/2607.00678",
    "codeUrls": [
      "https://github.com/amap-cvlab/ABot-Manipulation"
    ],
    "projectUrl": "https://amap-cvlab.github.io/ABot-Manipulation/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2607.00678",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "视觉规划与IDM",
      "导航",
      "泛化与动作对齐"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2607.04434",
    "title": "RoboDojo: A Unified Sim-and-Real Benchmark for Comprehensive Evaluation of Generalist Robot Manipulation Policies",
    "authors": "Tianxing Chen; Yue Chen; Zixuan Li; Junyuan Tang; Kailun Su; Haoran Lu; Weijie Wan; Baijun Chen; Songling Liu; Haowen Yan; Honghao Su; Zhiyang Dou; Kaixuan Wang; Dandan Zhang; Yunze Liu; Yan Qin; Qiwei Liang; Qiwei Wu; Zijian Lin; Wenwei Lin; Yuran Wang; Minghua He; Tianshu Wu; Ruihai Wu; Jingquan Zhou; Kai-Chong Lei; Haibao Yu; Yuanfeng Ji; Weiyang Jin; Guanyu Lin; Xiaofan Li; Qi Xiong; Renjing Xu; Zhongyu Li; Wenhao Chai; Enze Xie; Ziwei Wang; Yao Mu; Hao Dong; Wojciech Matusik; Mingyu Ding; Wenbo Ding; Ping Luo; Masayoshi Tomizuka",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2026-07-05",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "robodojo",
    "arxivUrl": "https://arxiv.org/abs/2607.04434",
    "codeUrls": [
      "https://github.com/RoboDojo-Benchmark/RoboDojo"
    ],
    "projectUrl": "https://robodojo-benchmark.com/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2607.04434",
    "pdfUrl": "https://arxiv.org/pdf/2607.04434",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{robodojo,\n  title={{RoboDojo}: A Unified Sim-and-Real Benchmark for Comprehensive Evaluation of Generalist Robot Manipulation Policies},\n  author={Chen, Tianxing and Chen, Yue and Li, Zixuan and Tang, Junyuan and Su, Kailun and Wan, Weijie and Chen, Baijun and Lu, Haoran and Yan, Haowen and Su, Honghao and others},\n  journal={arXiv:2607.04434},\n  year={2026}\n}",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "机器人操作基准",
      "仿真到真实评测",
      "评测协议与诊断"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2607.04265",
    "title": "HALO-WA: Hybrid-Attention Latent-Guided Online Reinforcement Learning for World-Action Models",
    "authors": "Ye, Angen; Ke, Weijie; Wang, Xiaofeng; Chen, Xinze; Ni, Chaojun; Zhao, Guosheng; Wang, Boyuan; Zhu, Zheng; Xie, Junjie; Zhang, Dapeng",
    "affiliations": "1 Institute of Automation, Chinese Academy of Sciences, Beijing, China; 2 School of Artificial Intelligence, University of Chinese Academy of Sciences, Beijing, China; 4 Tsinghua University, Beijing, China",
    "contribution": "We propose HALO-WA, a hybrid-attention latent-guided online reinforcement learning (RL) framework for WA models, which leverages latent features and action priors from the WA generation process through a lightweight actor-critic adapter to enable fast online adaptation to real deployment errors.",
    "abstract": "World-action (WA) models can generate long-horizon action chunks for general-purpose robotic manipulation, but they remain vulnerable to calibration, perception, and contact-dynamics errors in real-world precision tasks, often failing in the final few millimeters of alignment or insertion. We propose HALO-WA, a hybrid-attention latent-guided online reinforcement learning (RL) framework for WA models, which leverages latent features and action priors from the WA generation process through a lightweight actor-critic adapter to enable fast online adaptation to real deployment errors. HALO-WA introduces a hybrid-attention structure that preserves the temporal consistency of action chunks while reading task-relevant information from WA latents conditioned on visual context and end-stage correction requirements, thereby producing refined action chunks. We validate HALO-WA on four real-world precision manipulation tasks, where it improves the average success rate from 26.4\\% for WA-base to 87.1\\%, outperforming the strongest baseline by 19.2 percentage points while requiring only 45--75 minutes of online training per task. To facilitate reproducibility, we further conduct supplementary simulation experiments in RoboTwin and release the code at https://github.com/YeanRoot/HALO-WA.",
    "submittedDate": "2026-07-05",
    "primaryCategory": "WAM + RL",
    "secondaryCategories": [
      "Latent / Representation WAM",
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "HALOWA",
    "arxivUrl": "https://arxiv.org/abs/2607.04265",
    "codeUrls": [
      "https://github.com/YeanRoot/HALO-WA"
    ],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2607.04265",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "策略后训练与WM-RL",
      "泛化与动作对齐"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2606.16533",
    "title": "Kairos: A Regret-Aware Native World-Action Model Stack for Physical AI",
    "authors": "Team, Kairos; Wang, Fei; You, Shan; Zhang, Qiming; Huang, Tao; Fu, Zuoyi; Zheng, Zhisheng; Xi, Yunlong; Lv, Feng; Wu, Xiaoming; Liu, Zeyu; Wan, Cong; Li, Pu; Yang, Ruiqing; Li, Xiaoou; Wang, Wei; Zhu, Kangkang; Zhang, Yuwei; Fu, Shi; Zhang, Zheng; Wu, Xiaoning; Fan, Xuzeng; Tao, Dacheng; Wang, Xiaogang",
    "affiliations": "Not identified",
    "contribution": "We introduce \\textbf{Kairos}, a regret-aware native world-action model stack for Physical AI. Experiments on embodied world-model benchmarks, world-action benchmarks, long-horizon generation, and inference-efficiency evaluation show that Kairos achieves superior performance while offering a favorable efficiency to capability trade-off.",
    "abstract": "We introduce \\textbf{Kairos}, a regret-aware native world-action model stack for Physical AI. Kairos is motivated by the view that a physical world model should not aim to fully simulate all future pixels, but should learn and maintain the information most relevant to embodiment control: object state, spatial relations, contact conditions, task progress, action consequences, failure boundaries, and deployment uncertainty. Kairos establishes three model-side prerequisites toward this goal. First, it \\textbf{learns} control-relevant information through a \\textbf{Cross-Embodiment Data Curriculum}, which organizes open-world videos, human behavioral data, and robot interactions into an intervention-strength progression from passive physical observation to intentional behavior and embodied action grounding. Second, it \\textbf{maintains} control-sufficient states through a unified \\textbf{understanding, generation, and prediction architecture} equipped with \\textbf{Hybrid Linear Temporal Attention}, where local, mid-range, and global temporal pathways support multi-timescale state maintenance under efficient inference. Third, it \\textbf{deploys} these states through a \\textbf{Deployment-Aware System Co-Design}, treating latency, memory footprint, and hardware compatibility as first-order constraints for future observation, action, and feedback loops. Experiments on embodied world-model benchmarks, world-action benchmarks, long-horizon generation, and inference-efficiency evaluation show that Kairos achieves superior performance while offering a favorable efficiency to capability trade-off.",
    "submittedDate": "2026-07-03",
    "primaryCategory": "Memory WAM",
    "secondaryCategories": [
      "Evaluation / Survey / Theory",
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "Kairos",
    "arxivUrl": "https://arxiv.org/abs/2606.16533",
    "codeUrls": [
      "https://github.com/kairos-agi/kairos"
    ],
    "projectUrl": "https://kairos.acerobotics.com",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2606.16533",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "联合视频动作建模",
      "记忆与长时序",
      "高效推理与实时控制"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "联合预测",
    "quadrant": "Q3 · Dual-system × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2607.02503",
    "title": "VT-WAM: Visual-Tactile World Action Model for Contact-Rich Manipulation",
    "authors": "Tian, Shuai; Zheng, Yupeng; Zheng, Yuhang; Gu, Songen; Zang, Yujie; Qin, Yuxing; Li, Weize; Li, Haoran; Ding, Wenchao; Zhao, Dongbin",
    "affiliations": "1 SKL-MAIS, Institute of Automation, Chinese Academy of Sciences; 2 School of Artificial Intelligence, University of Chinese Academy of Sciences; 3 TARS Robotics; 4 National University of Singapore; 5 Fudan University",
    "contribution": "In this paper, we introduce VT-WAM, a Visual-Tactile World Action Model that jointly learns future visual prediction, tactile deformation prediction, and action prediction within a unified flow matching framework. Across six real-world contact-rich manipulation tasks, VT-WAM achieves a 71.67% average success rate, outperforming Fast-WAM by 26.67% and OmniVTLA by 35.84%.",
    "abstract": "Contact-rich manipulation requires policies to react to local deformation, pressure, slip, and friction, yet these cues are temporally sparse and often invisible in visual observations. Existing visual-tactile policies usually feed tactile observations directly into action prediction, but rarely model tactile deformation dynamics during action generation. In this paper, we introduce VT-WAM, a Visual-Tactile World Action Model that jointly learns future visual prediction, tactile deformation prediction, and action prediction within a unified flow matching framework. In particular, VT-WAM introduces (1) Asymmetric Mixture-of-Transformers (MoT) attention to bridge a first-frame visual anchor with temporal tactile dynamics, and (2) contact-gated Action-Visual-Tactile Attention Guidance (AVTAG) to encourage action queries to rely on tactile evidence during contact phases. Across six real-world contact-rich manipulation tasks, VT-WAM achieves a 71.67% average success rate, outperforming Fast-WAM by 26.67% and OmniVTLA by 35.84%. Ablations demonstrate that modeling tactile deformation dynamics and guiding contact-phase tactile attention are both important for contact-rich tasks. Project website: https://vt-wam.github.io/.",
    "submittedDate": "2026-07-02",
    "primaryCategory": "Multimodal / Tactile WAM",
    "secondaryCategories": [
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "VTWAM",
    "arxivUrl": "https://arxiv.org/abs/2607.02503",
    "codeUrls": [],
    "projectUrl": "https://vt-wam.github.io/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2607.02503",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "多模态触觉音频",
      "联合视频动作建模",
      "高效推理与实时控制"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "联合预测",
    "quadrant": "Q3 · Dual-system × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2607.02195",
    "title": "Bridge-WA: Predicting Where and How the World Changes for Robotic Action",
    "authors": "Bai, Yongjie; Wang, Hanting; Dai, Mingtong; Zhong, Qijun; Liu, Yang; Lin, Liang",
    "affiliations": "1 Sun Yat-sen University; 2 Pengcheng Laboratory; 3 Shenzhen Institutes of Advanced Technology, Chinese Academy of Sciences; 4 X-Era AI Lab",
    "contribution": "We present Bridge-WA, a lightweight world-action framework that distills a frozen future-change teacher into three compact priors: future tokens for intended outcomes, change maps for intervention support, and motion-flow maps for local transition direction.",
    "abstract": "General-purpose vision-language-action models benefit from large vision-language priors, but effective manipulation also requires anticipating action-relevant scene changes. Existing world-action models often rely on large generative world models or dense future rollouts, which are expensive and spend capacity on visual details weakly coupled to control. We present Bridge-WA, a lightweight world-action framework that distills a frozen future-change teacher into three compact priors: future tokens for intended outcomes, change maps for intervention support, and motion-flow maps for local transition direction. A WorldBridge conditions the action transformer on these priors through multi-source attention memories and spatial-temporal biases, while the teacher model is removed at inference. Across VLABench, RoboTwin2.0, LIBERO-Plus and real-robot evaluations, Bridge-WA improves task success, progress, and robustness, with particularly clear gains under out-of-distribution visual shifts. By focusing action generation on where and how the scene will change, Bridge-WA suppresses nuisance appearance factors such as background, lighting, and distractors, leading to better generalization without deployment-time dense future-image generation. Code and visualizations are available at: https://hcplab-sysu.github.io/BRIDGE-WA .",
    "submittedDate": "2026-07-02",
    "primaryCategory": "Memory WAM",
    "secondaryCategories": [
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "BridgeWA",
    "arxivUrl": "https://arxiv.org/abs/2607.02195",
    "codeUrls": [
      "https://github.com/HCPLab-SYSU/BRIDGE-WA"
    ],
    "projectUrl": "https://hcplab-sysu.github.io/BRIDGE-WA",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2607.02195",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "潜空间预测与JEPA",
      "视觉规划与IDM",
      "高效推理与实时控制"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2606.31101",
    "title": "Efficient Sim-to-Real Transfer of World-Action Models from Synthetic Priors",
    "authors": "Wang, Zixing; Sivakumar, Kausik; Shang, Jinghuan; Hu, Yafei; Xie, Zhaoming; Gong, Ran; Zhang, Xiaohan; Schmeckpeper, Karl",
    "affiliations": "Purdue University; Robotics and AI Institute",
    "contribution": "To this end, we build upon Cosmos Policy, a video diffusion model adapted for visuomotor control. We evaluate our approach across object lifting, drawer opening, and pick-and-place tasks using ∼800{\\sim}800 synthetic demonstrations per task and no real demonstrations.",
    "abstract": "Bridging the sim-to-real gap is a core challenge in deploying learned manipulation policies. Sim-to-real learning is attractive because it can replace expensive real robot demonstrations with scalable synthetic data, yet world-action models have not previously been shown to transfer from simulation to real robotic manipulation. We study whether a world-action model can be trained from synthetic priors and deployed zero-shot in the real world. To this end, we build upon Cosmos Policy, a video diffusion model adapted for visuomotor control. We construct simulation environments with extensive domain randomization and generate demonstrations using the AnyTask motion planning pipeline. We evaluate our approach across object lifting, drawer opening, and pick-and-place tasks using ∼800{\\sim}800 synthetic demonstrations per task and no real demonstrations. When deployed zero-shot on a Franka Robot, our policy attains a 35\\% average success rate. To our knowledge, this represents the first successful sim-to-real transfer of a world-action model for robotic manipulation.",
    "submittedDate": "2026-06-29",
    "primaryCategory": "Real-Time / Efficient WAM",
    "secondaryCategories": [],
    "bibtexKey": "ESRTSP",
    "arxivUrl": "https://arxiv.org/abs/2606.31101",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2606.31101",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "泛化与动作对齐",
      "联合视频动作建模"
    ],
    "architecture": "One Model",
    "predictionParadigm": "联合预测",
    "quadrant": "Q1 · One Model × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2606.30367",
    "title": "FutureNav: Unified World-Action Modeling for Vision-and-Language Navigation",
    "authors": "Zhang, Lingfeng; Gong, Zeying; Hao, Xiaoshuai; Fu, Haoxiang; Zhang, Qiang; Zhou, Mingliang; Ye, Hangjun; Liang, Xiaojun; Liang, Junwei; Ding, Wenbo",
    "affiliations": "1Tsinghua University2Pengcheng Laboratory 3The Hong Kong University of Science and Technology; (Guangzhou) 4Xiaomi EV 5National University of Singapore",
    "contribution": "We introduce FutureNav, a VLM-based unified world-action modeling framework for vision-and-language navigation. Extensive experiments show that, with only a 4B-scale backbone, FutureNav achieves state-of-the-art performance on multiple VLN benchmarks and substantially outperforms prior VLN methods, paving the way toward future world-action models for VLN.",
    "abstract": "Vision-and-language navigation (VLN) in continuous environments requires an agent to ground instructions in egocentric observations while maintaining spatial understanding across long action sequences. Recent navigation foundation models have shown strong progress by scaling vision-language models, but they often learn navigation primarily as direct action generation, without explicitly modeling world states or predicting their future evolution. We introduce FutureNav, a VLM-based unified world-action modeling framework for vision-and-language navigation. Specifically, FutureNav jointly encodes text, visual, and spatial features and feeds them into the LLM, and optimizes four objectives for simultaneous world and action modeling: an action policy objective for navigation action prediction, inverse and forward dynamics objectives for modeling state transitions, and a future generation objective for predicting future spatial states. This unified architecture strengthens action prediction while explicitly modeling the world, without sacrificing inference speed. Extensive experiments show that, with only a 4B-scale backbone, FutureNav achieves state-of-the-art performance on multiple VLN benchmarks and substantially outperforms prior VLN methods, paving the way toward future world-action models for VLN. We will release the code and models to support future research.",
    "submittedDate": "2026-06-29",
    "primaryCategory": "Navigation / Driving / Domain WAM",
    "secondaryCategories": [],
    "bibtexKey": "FutureNav",
    "arxivUrl": "https://arxiv.org/abs/2606.30367",
    "codeUrls": [
      "https://github.com/linglingxiansen/FutureNav"
    ],
    "projectUrl": "https://linglingxiansen.github.io/FutureNav/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2606.30367",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "导航",
      "潜空间预测与JEPA"
    ],
    "architecture": "One Model",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2606.29908",
    "title": "Pondering the Way: Spatial-perceiving World Action Model for Embodied Navigation",
    "authors": "Chen, Hong; Liu, Daqi; Zhang, Zehan; Wang, Haiguang; Lu, Tianhao; Yan, Longfei; Sun, Haiyang; Li, Fangzhen; Xie, Hongwei; Wang, Bing; Chen, Guang; Ye, Hangjun; Tan, Yihua",
    "affiliations": "Huazhong University of Science and Technology, Wuhan, China hongc, yhtan @hust.edu.cn Xiaomi EV, Beijing, China Zhejiang University, Hangzhou, China",
    "contribution": "To address these issues, we propose SWAM (Spatial-perceiving World Action Model), a task-centric joint observation-action generation framework. Extensive experiments show that SWAM significantly outperforms state-of-the-art two-stage planners in success rate, trajectory accuracy, and inference efficiency, while demonstrating robust zero-shot generalization to unseen environments.",
    "abstract": "Existing world model-based planners for visual navigation typically follow a verification-centric paradigm, decoupling goal intent from trajectory synthesis. This approach suffers from candidate dependence, heavy computational overhead, and inconsistencies between sampled actions and predicted visuals. To address these issues, we propose SWAM (Spatial-perceiving World Action Model), a task-centric joint observation-action generation framework. Given start and goal RGB observations, SWAM performs single-pass inference to simultaneously generate intermediate RGB-D sequences and corresponding action trajectories, promoting goal-consistent trajectory generation and improved spatial feasibility. While SWAM leverages depth pseudo-labels during training to internalize spatial priors, it requires only monocular RGB input at inference time. We further introduce a visual-guided action refinement module and a trajectory-scale regularization loss to enforce fine-grained alignment between motion and visual cues while stabilizing predictions across varying distances. Extensive experiments show that SWAM significantly outperforms state-of-the-art two-stage planners in success rate, trajectory accuracy, and inference efficiency, while demonstrating robust zero-shot generalization to unseen environments.",
    "submittedDate": "2026-06-29",
    "primaryCategory": "Navigation / Driving / Domain WAM",
    "secondaryCategories": [
      "3D/4D WAM",
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "SWAM",
    "arxivUrl": "https://arxiv.org/abs/2606.29908",
    "codeUrls": [],
    "projectUrl": null,
    "venue": "ECCV 2026",
    "paperUrl": "https://arxiv.org/abs/2606.29908",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "导航",
      "联合视频动作建模",
      "三维多视角建模"
    ],
    "architecture": "One Model",
    "predictionParadigm": "联合预测",
    "quadrant": "Q1 · One Model × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2606.27374",
    "title": "World Action Models Enable Continual Imitation Learning with Recurrent Generative Replays",
    "authors": "Govind, Manish Kumar; Reilly, Dominick; Patel, Smit; Le, Hieu; Das, Srijan",
    "affiliations": "Department of Computer Science; University of North Carolina at Charlotte, United States",
    "contribution": "We build on this generative capability to propose Recurrent Generative Replay (REGEN), a continual imitation learning framework that synthesizes pseudo-replay trajectories, enabling a robot policy to rehearse previously learned tasks without storing their original human demonstrations.",
    "abstract": "Going beyond predicting robot actions, World Action Models (WAMs) can also generate future visual observations. We build on this generative capability to propose Recurrent Generative Replay (REGEN), a continual imitation learning framework that synthesizes pseudo-replay trajectories, enabling a robot policy to rehearse previously learned tasks without storing their original human demonstrations. During continual adaptation, REGEN recursively queries the WAM to synthesize pseudo-replay trajectories conditioned only on prior task instructions and current-task observations. Experiments in both simulation and real-world manipulation settings show that REGEN reduces catastrophic forgetting by up to 50%50\\% relative to sequential fine-tuning, while approaching the performance of privileged experience replay methods that require access to real replay data. Finally, we analyze the factors limiting generated replay, identifying long-horizon visual degradation and action-observation inconsistency as the primary bottlenecks. Our results establish WAMs as a promising foundation for continual robot learning without stored demonstrations.",
    "submittedDate": "2026-06-25",
    "primaryCategory": "Memory WAM",
    "secondaryCategories": [],
    "bibtexKey": "REGEN",
    "arxivUrl": "https://arxiv.org/abs/2606.27374",
    "codeUrls": [
      "https://github.com/ManishGovind/REGEN"
    ],
    "projectUrl": "https://manishgovind.github.io/REGEN/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2606.27374",
    "pdfUrl": "https://arxiv.org/pdf/2606.27374",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{REGEN,\n  title={World Action Models Enable Continual Imitation Learning with Recurrent Generative Replays},\n  author={Govind, Manish Kumar and Reilly, Dominick and Patel, Smit and Le, Hieu and Das, Srijan},\n  journal={arXiv:2606.27374},\n  year={2026}\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "联合视频动作建模",
      "记忆与长时序",
      "泛化与动作对齐"
    ],
    "architecture": "One Model",
    "predictionParadigm": "联合预测",
    "quadrant": "Q1 · One Model × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2606.18375",
    "title": "PAIWorld: A 3D-Consistent World Foundation Model for Robotic Manipulation",
    "authors": "Yuhang Huang; Xuan Lv; Junyan Xu; Zhiyuan Yu; Jiazhao Zhang; Ruizhen Hu; Wancheng Feng; Shilong Zou; Hewen Xiao; Ziqiao Zhou; Kaiyun Huang; Zhiyu Peng; Juzhan Xu; Hang Zhao; Chenyang Zhu; Renjiao Yi; Yifei Huang; Douhui Wu; Yan Zhang; Kexu Cheng; Chunhe Song; Yunzhi Xue; Xiuhong Zhang; Leitao Guo; Yunji Chen; Bin Wu; Haibin Yu; Kai Xu",
    "affiliations": "Institute of AI for Industries, Chinese Academy of Sciences",
    "contribution": "To address this, we present PAIWorld, a framework that augments diffusion-transformer world models via three core components: (1) Geometry-Aware Cross-View Attention blocks that establish an explicit pathway across views, (2) Geometric Rotary Position Embedding that encodes camera ray directions and extrinsic poses into the attention mechanism, and (3) Latent 3D-REPA, which distills 3D-aware features from frozen 3D foundation models to ensure 3D consistency. Built upon a DiT-based world foundation model, PAIWorld achieves state-of-the-art multi-view 3D consistency on robotic manipulation benchmarks, ranking 1st on the WorldArena leaderboard and 2nd on the AgiBot-Challenge2026 leaderboard, wh…",
    "abstract": "World foundation models (WFMs) are powerful simulators, yet they predominantly operate in a single-view setting and lack the multi-view 3D consistency required for robotic manipulation. While robotic systems rely on multiple cameras (egocentric, eye-to-hand, and wrist-mounted) for policy learning, current multi-view world models simply concatenate view tokens without explicit geometric reasoning. This causes cross-view object drift, depth inconsistency, and texture misalignment. We trace these failures to two deficiencies: the absence of an explicit inter-view communication mechanism and the lack of a 3D geometric prior. We argue that resolving both simultaneously is necessary and sufficient. To address this, we present PAIWorld, a framework that augments diffusion-transformer world models via three core components: (1) Geometry-Aware Cross-View Attention blocks that establish an explicit pathway across views, (2) Geometric Rotary Position Embedding that encodes camera ray directions and extrinsic poses into the attention mechanism, and (3) Latent 3D-REPA, which distills 3D-aware features from frozen 3D foundation models to ensure 3D consistency. Built upon a DiT-based world foundation model, PAIWorld achieves state-of-the-art multi-view 3D consistency on robotic manipulation benchmarks, ranking 1st on the WorldArena leaderboard and 2nd on the AgiBot-Challenge2026 leaderboard, while enabling downstream applications such as model-based planning, world action models, and multi-view policy post-training.",
    "submittedDate": "2026-06-23",
    "primaryCategory": "3D/4D WAM",
    "secondaryCategories": [
      "Latent / Representation WAM"
    ],
    "bibtexKey": "PAIWorld",
    "arxivUrl": "https://arxiv.org/abs/2606.18375",
    "codeUrls": [],
    "projectUrl": "https://guhuangai.github.io/PAIWorld-Proj/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2606.18375",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "神经世界模拟器"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2606.02800",
    "title": "Cosmos 3: Omnimodal World Models for Physical AI",
    "authors": "NVIDIA; Aditi; Niket Agarwal; Arslan Ali; Jon Allen; Martin Antolini; Adeline Aubame; Alisson Azzolini; Junjie Bai; Maciej Bala; Yogesh Balaji; Josh Bapst; Aarti Basant; Mukesh Beladiya; Mohammad Qazim Bhat; Zaid Pervaiz Bhat; Dan Blick; Vanni Brighella; Han Cai; Tiffany Cai; Eric Cameracci; Jiaxin Cao; Yulong Cao; Mark Carlson; Carlos Casanova; Ting-Yun Chang; Yan Chang; Yu-Wei Chao; Prithvijit Chattopadhyay; Roshan Chaudhari; Chieh-Yun Chen; Junyu Chen; Ke Chen; Qizhi Chen; Wenkai Chen; Xiaotong Chen; Yu Chen; An-Chieh Cheng; Click Cheng; Xiu Chia; Jeana Choi; Chaeyeon Chung; Wenyan Cong; Yin Cui; Magdalena Dadela; Nalin Dadhich; Wenliang Dai; Joyjit Daw; Alperen Degirmenci; Rodrigo Vieira Del Monte; Robert Denomme; Sameer Dharur; Marco Di Lucca; Ke Ding; Wenhao Ding; Yifan Ding; Yuzhu Dong; Nicole Drumheller; Yilun Du; Aigul Dzhumamuratova; Aleksandr Efitorov; Hamid Eghbalzadeh; Naomi Eigbe; Imad El Hanafi; Hassan Eslami; Benedikt Falk; Jiaojiao Fan; Jim Fan; Amol Fasale; Sergiy Fefilatyev; Liang Feng; Francesco Ferroni; Sanja Fidler; Xiao Fu; Vikram Fugro; Prashant Gaikwad; TJ Galda; Katelyn Gao; Yihuai Gao; Wenhang Ge; Sreyan Ghosh; Arushi Goel; Vivek Goel; Akash Gokul; Rama Govindaraju; Jinwei Gu; Miguel Guerrero; Elfie Guo; Aryaman Gupta; Siddharth Gururani; Hugo Hadfield; Song Han; Ankur Handa; Zekun Hao; Mohammad Harrim; Ali Hassani; Nathan Hayes-Roth; Yufan He; Chris Helvig; Cyrus Hogg; Madison Huang; Michael Huang; Sophia Huang; Yufan Huang; Jacob Huffman; DeLesley Hutchins; Suneel Indupuru; Boris Ivanovic; Arihant Jain; Joel Jang; Ryan Ji; Yanan Jian; Dongfu Jiang; Jingyi Jin; Atharva Joshi; Nikhilesh Joshi; Pranjali Joshi; Andy Ju; Jaehun Jung; Weiwei Kang; Scott Kassekert; Jan Kautz; Ashna Khetan; Julia Kiczka; Slawek Kierat; Gwanghyun Kim; Kuno Kim; Sunny Kim; Kezhi Kong; Xin Kong; Zhifeng Kong; Tomasz Kornuta; Egor Krivov; Hui Kuang; Saurav Kumar; Chia-Wen Kuo; George Kurian; Wojciech Kutak; JF Lafleche; Himangshu Lahkar; Omar Laymoun; Jayjun Lee; Sanggil Lee; Gabriele Leone; Boyi Li; Freya Li; Jiajun Li; Jinfeng Li; Ling Li; Pengcheng Li; Shangru Li; Tingle Li; Xiaolong Li; Xuan Li; Zhaoshuo Li; Zhiqi Li; Hao Liang; Maosheng Liao; Chen-Hsuan Lin; Tsung-Yi Lin; Ming-Yu Liu; Sifei Liu; Zihan Liu; Hai Loc Lu; Xiangyu Lu; Alice Luo; Ruipu Luo; Wenjie Luo; Jiangran Lyu; Martin Ding Ma; Nic Ma; Qianli Ma; Dawid Majchrowski; Louis Marcoux; Miguel Martin; Qing Miao; Ashkan Mirzaei; Shreyas Misra; Kaichun Mo; Durra Mohsin; Hyejin Moon; Pawel Morkisz; Saeid Motiian; Kirill Motkov; Seungjun Nah; Yashraj Narang; Deepak Narayanan; Thabang Ngazimbi; Julian Ouyang; Shubham Pachori; David Page; Yatian Pang; Sehwi Park; Mahesh Patekar; Mostofa Patwary; Marco Pavone; Trung Pham; Wei Ping; Soha Pouya; Shrimai Prabhumoye; Varun Praveen; Delin Qu; Hesam Rabeti; Morteza Ramezanali; Marilyn Reeb; Xuanchi Ren; Kristen Rumley; Wojciech Rymer; Jun Saito; Yeongho Seol; John Shao; Piyush Shekdar; Tianwei Shen; Humphrey Shi; Min Shi; Stella Shi; Kevin Shih; Mohammad Shoeybi; Mateusz Sieniawski; Shuran Song; Alexander Sotelo; Amir Sotoodeh; Sunil Srinivasa; Vignesh Srinivasakumar; Bartosz Stefaniak; Rahul Heinrich Steiger; Shangkun Sun; Jiaxiang Tang; Shitao Tang; Yangyang Tang; Yue Tang; Tolou Tavakkoli; Kayley Ting; Krzysztof Tomala; Wei-Cheng Tseng; Jibin Varghese; Sergei Vasilev; Thomas Volk; Raju Wagwani; Roger Waleffe; Andrew Z. Wang; Boxiang Wang; Haoxiang Wang; Qiao Wang; Shihao Wang; Shijie Wang; Ting-Chun Wang; Yan Wang; Yu Wang; Rohit Watve; David Wehr; Fangyin Wei; Xinshuo Weng; Jay Zhangjie Wu; Kedi Wu; Hongchi Xia; Summer Xiao; Tianjun Xiao; Kevin Xie; Daguang Xu; Jiashu Xu; Mengyao Xu; Ruqing Xu; Xingqian Xu; Yao Xu; Dinghao Yang; Dong Yang; Hans Yang; Xiaodong Yang; Xuning Yang; Yichu Yang; Yurong You; Zhiding Yu; Hao Yuan; Simon Yuen; Xiaohui Zeng; Pengcuo Zeren; Cindy Zha; Haotian Zhang; Jenny Zhang; Jing Zhang; Liangkai Zhang; Paris Zhang; Shun Zhang; Xuanmeng Zhang; Zhizheng Zhang; Ann Zhao; Yilin Zhao; Yuliya Zhautouskaya; Charles Zhou; Fengzhe Zhou; Shilin Zhu; Yuke Zhu; Dima Zhylko; Artur Zolkowski",
    "affiliations": "NVIDIA Contributors and acknowledgments are listed in Appendix sec::contributors",
    "contribution": "We introduce Cosmos 3, a family of omnimodal world models designed to jointly process and generate language, image, video, audio, and action sequences within a unified mixture-of-transformers architecture. Our evaluation demonstrates that Cosmos 3 establishes a new state-of-the-art across a diverse suite of understanding and generation tasks, demonstrating omnimodal world models as scalable, general-purpose backbones for embodied agents.",
    "abstract": "We introduce Cosmos 3, a family of omnimodal world models designed to jointly process and generate language, image, video, audio, and action sequences within a unified mixture-of-transformers architecture. By supporting highly flexible input-output configurations, Cosmos 3 seamlessly unifies critical modalities for Physical AI -- effectively subsuming vision-language models, video generators, world simulators, and world-action models into a single framework. Our evaluation demonstrates that Cosmos 3 establishes a new state-of-the-art across a diverse suite of understanding and generation tasks, demonstrating omnimodal world models as scalable, general-purpose backbones for embodied agents. Our post-trained Cosmos 3 models were ranked as the best open-source Text-to-Image and Image-to-Video models by Artificial Analysis, and the best policy model by RoboArena at the time the technical report was written. To accelerate open research and deployment in Physical AI, we make our code, model checkpoints, curated synthetic datasets, and evaluation benchmark available under the Linux Foundation's OpenMDW-1.1 License at https://github.com/nvidia/cosmos and https://huggingface.co/collections/nvidia/cosmos3. The project website is available at https://research.nvidia.com/labs/cosmos-lab/cosmos3.",
    "submittedDate": "2026-06-23",
    "primaryCategory": "Multimodal / Tactile WAM",
    "secondaryCategories": [
      "Evaluation / Survey / Theory",
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "Cosmos3",
    "arxivUrl": "https://arxiv.org/abs/2606.02800",
    "codeUrls": [
      "https://github.com/NVIDIA/cosmos"
    ],
    "projectUrl": "https://research.nvidia.com/labs/cosmos-lab/cosmos3/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2606.02800",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "联合视频动作建模",
      "多模态触觉音频",
      "泛化与动作对齐"
    ],
    "architecture": "One Model",
    "predictionParadigm": "联合预测",
    "quadrant": "Q1 · One Model × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2606.23991",
    "title": "Critique of Agent Model",
    "authors": "Eric Xing; Mingkai Deng; Jinyu Hou",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2026-06-22",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "xing2026CAM",
    "arxivUrl": "https://arxiv.org/abs/2606.23991",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2606.23991",
    "pdfUrl": "https://arxiv.org/pdf/2606.23991",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{xing2026CAM,\n  title={Critique of Agent Model},\n  author={Xing, Eric and Deng, Mingkai and Hou, Jinyu},\n  journal={arXiv:2606.23991},\n  year={2026}\n}",
    "majorCategory": "Related resources",
    "subcategories": [
      "理论与规划",
      "综述与技术资源"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2606.23623",
    "title": "dVLA-RL: Reinforcement Learning over Denoising Trajectories for Discrete Diffusion Vision-Language-Action Models",
    "authors": "Wu, Yuhao; Liu, Yitian; Shen, Weijie; Han, Mishuo; Xu, Wenjie; Liang, Haotian; Liu, Zhongshan; Mao, Yinan; Xu, Lei; Guan, Xinping; Ying, Ru; Zheng, Ran; Sui, Wei; Yang, Xiaokang; Ding, Wenbo; Mu, Yao",
    "affiliations": "Yuhao Wu 1,2*, Yitian Liu 1,3*, Weijie Shen 1,4 Co-first authors, equal contribution. Corresponding authors. Project co-leaders. This work was completed by Yuhao Wu during his internship at ScaleLab at SJTU. Wei Sui at D-Robotics and Ru Ying at Baidu AI Cloud are co-leaders of this project., Mishuo Han 1, Wenjie Xu 3, Haotian Liang 5 3pt; 1 Shanghai Jiao Tong University 1pt; 2 Tsinghua Shenzhen International Graduate School, Tsinghua University 1pt; 3 Baidu AI Cloud 4 D-Robotics 1pt 5 Shanghai AI Laboratory",
    "contribution": "To solve this problem, we propose \\textbf{dVLA-RL}, shifting the learning objective from the marginal action probability to the joint probability of the sampled generation path. Leveraging this intrinsic fexibility, we introduce a unified step scheduling approach for complex multi-task learning, tailoring denoising steps to specific task complexities to maximize both success rates and computational effciency.",
    "abstract": "Vision-Language-Action (VLA) models have established a powerful paradigm for generalist robotic manipulation by grounding control into the semantic reasoning of VLMs. Prevailing architectures typically model actions continuously via diffusion or flow processes, or discretely through either autoregressive generation or parallel decoding. Recently, Discrete Diffusion VLAs (dVLAs) have emerged as a distinct alternative, unifying vision, language, and action into a single discrete token space via masked generative modeling. While combining iterative refinement with unified representations, its training has thus far been restricted to Supervised Fine-Tuning (SFT), leaving the potential of Reinforcement Learning (RL) for further policy refinement largely unexplored. A fundamental challenge in RL for dVLAs is that the marginal probability of the final action generated by dVLAs remains intractable. To solve this problem, we propose \\textbf{dVLA-RL}, shifting the learning objective from the marginal action probability to the joint probability of the sampled generation path. Specifically, by modeling the denoising process as a Markov Decision Process (MDP), we mathematically formulate this path probability as a product of step-wise transitions. This trajectory-level objective provides a unified formulation that natively accommodates variable denoising steps. Leveraging this intrinsic fexibility, we introduce a unified step scheduling approach for complex multi-task learning, tailoring denoising steps to specific task complexities to maximize both success rates and computational effciency. Extensive evaluations demonstrate that our approach achieves a success rate of \\textbf{99.7\\%} on LIBERO. Furthermore, it establishes strong VLA-based results on RoboTwin 2.0 by delivering a \\textbf{30.6\\%} improvement over the SFT baseline, remaining competitive with strong World-Action Model baselines.",
    "submittedDate": "2026-06-22",
    "primaryCategory": "WAM + RL",
    "secondaryCategories": [],
    "bibtexKey": "dVLARL",
    "arxivUrl": "https://arxiv.org/abs/2606.23623",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2606.23623",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "VLA",
    "subcategories": [
      "扩散与流匹配VLA",
      "VLA后训练与数据增强"
    ],
    "architecture": "One Model",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2606.23574",
    "title": "A Watermark for Vision-Language-Action and World Action Models",
    "authors": "Liu, Yule; Liu, Shuai; Wei, Jiaheng; He, Xinlei",
    "affiliations": "Hong Kong University of Science and Technology (Guangzhou); Wuhan University; Xi'an Jiao Tong University",
    "contribution": "To address this, we propose the \\emph{keyed latent-provenance verification} method, which fingerprints the policy through the seed of the Gaussian noise vector that the models draw before generation.",
    "abstract": "Vision-language-action (VLA) models and world-action models (WAM) are the generative models now driving general-purpose robot control, turning raw camera input directly into motor commands. They are increasingly deployed as black-box services, where a partner runs the policy through an interface while the owner keeps the weights private. Training such a model takes proprietary data and heavy computational power, making the deployed model itself a valuable intellectual property. To address this, we propose the \\emph{keyed latent-provenance verification} method, which fingerprints the policy through the seed of the Gaussian noise vector that the models draw before generation. At the injection stage, the owner swaps this seed for a keyed one with the same distribution as ordinary noise, so the fingerprinted actions are statistically identical to those of an ordinary run and an adversary watching the output finds no signal to detect or remove. At the verification stage, the owner runs the suspect model under authorized access and records the action channels the robot executes, a partial and possibly post-processed view of the policy's output. From this view, the verifier recovers the seed by gradient-based maximum a posteriori (MAP) optimization, tests it for the secret key to score each rollout, and aggregates these scores into a single decision on whether the suspect model belongs to the owner. We evaluate the method on two representative models across two robot suites. The experiments cover detection of the fingerprint, identification of which of several keys a suspect carries, robustness to a range of attacks, and an analysis of why the design works. Across both models, the fingerprint can be detected reliably with little change to task performance, and it remains detectable under output-side removal attacks and weight-level edits.",
    "submittedDate": "2026-06-22",
    "primaryCategory": "Latent / Representation WAM",
    "secondaryCategories": [
      "Evaluation / Survey / Theory",
      "Navigation / Driving / Domain WAM"
    ],
    "bibtexKey": "WVL",
    "arxivUrl": "https://arxiv.org/abs/2606.23574",
    "codeUrls": [
      "https://github.com/Y-L-LIU/keyed-latent-watermarking-vla"
    ],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2606.23574",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "Related resources",
    "subcategories": [
      "评测协议与诊断"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2606.22966",
    "title": "Attacking the Trusted Imagination: Oracle-Level Integrity Attacks on Imagine-then-Act World Models",
    "authors": "Chen, Linghan; Ji, Kaiyan; Guo, Minyu",
    "affiliations": "Adelaide University",
    "contribution": "Many recent vision-language-action (VLA) policies adopt an imagine-then-act design. A world-action model (WAM) first imagines a short future as a latent trajectory z~, on which the action is then conditioned.",
    "abstract": "Many recent vision-language-action (VLA) policies adopt an imagine-then-act design. A world-action model (WAM) first imagines a short future as a latent trajectory z~, on which the action is then conditioned. We identify this trusted imagination, rather than the reactive policy, as the exposed attack surface. A downstream oracle, such as a safety gate, a visual model-predictive-control (MPC) planner, or an imagine-then-check verifier, consumes z~ as a prediction of the future. The robustness of the policy therefore does not entail the robustness of systems that rely on the WAM. The underlying phenomenon is an asymmetry. Corrupting the imagination is easy, since it requires only displacing z~ from its natural-future manifold. Steering it precisely is hard, since it must reach a specified on-manifold target. We adopt a capability-based threat model with an L-infinity-bounded observation perturbation. The attacker applies projected gradient descent through the fully differentiable observation-to-imagination map. The same off-manifold property motivates a parameter-free denoiser detector. We evaluate three targets: RynnVLA-002, LingBot-VA, and LaDi-WM. Untargeted corruption is roughly 60x stronger than random and is detected at AUC 1.0. Targeted control remains bounded. An adaptive attacker evades detection only by forgoing corruption. The reactive policy remains robust to corrupted imagination. A native imagination-driven MPC, however, exhibits the first adversary-specific task failure (at epsilon=0.01, success 0.70 versus 0.05; Fisher p < 10^-4).",
    "submittedDate": "2026-06-22",
    "primaryCategory": "Latent / Representation WAM",
    "secondaryCategories": [],
    "bibtexKey": "MPC",
    "arxivUrl": "https://arxiv.org/abs/2606.22966",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2606.22966",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "Related resources",
    "subcategories": [
      "评测协议与诊断"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2606.17046",
    "title": "Geometric Action Model for Robot Policy Learning",
    "authors": "Han, Jisang; Jeon, Seonghu; Jung, Jaewoo; Zurbrügg, René; An, Honggyu; Portela, Tifanny; Hutter, Marco; Pollefeys, Marc; Kim, Seungryong; Hong, Sunghwan",
    "affiliations": "1 KAIST AI 2 ETH Zurich 3 ETH AI Center",
    "contribution": "We propose the Geometric Action Model (GAM), a language-conditioned manipulation policy that directly repurposes a pretrained geometric foundation model (GFM) as a shared substrate for perception, temporal prediction, and action decoding.",
    "abstract": "Generalist robot policies must follow user instructions while reasoning about how objects, cameras, and robot actions interact in the 3D physical world. Recent vision-language-action models (VLAs) and video world-action models (WAMs) inherit strong semantic or temporal priors from large-scale foundation models, but they still operate primarily on 2D image frames or 2D-derived latent spaces, leaving implicit the 3D geometry required for contact-rich manipulation. We propose the Geometric Action Model (GAM), a language-conditioned manipulation policy that directly repurposes a pretrained geometric foundation model (GFM) as a shared substrate for perception, temporal prediction, and action decoding. GAM splits the GFM at an intermediate layer: the shallow layers serve as an observation encoder, and a causal future predictor inserted at the split layer forecasts future latent tokens conditioned on language, proprioception, and action history. The predicted future tokens are then routed through the remaining GFM blocks for feature propagation and decoding, allowing a single backbone to produce both future geometry and actions. This design equips the GFM with language-conditioned temporal world modeling through minimal architectural modification while preserving its rich geometric priors. Across a broad suite of simulation and real-robot manipulation benchmarks, GAM is more accurate, more robust, faster, and lighter than current foundation-model-scale baselines.",
    "submittedDate": "2026-06-22",
    "primaryCategory": "3D/4D WAM",
    "secondaryCategories": [
      "Latent / Representation WAM",
      "Multimodal / Tactile WAM"
    ],
    "bibtexKey": "GAM",
    "arxivUrl": "https://arxiv.org/abs/2606.17046",
    "codeUrls": [
      "https://github.com/cvlab-kaist/Geometric-Action-Model"
    ],
    "projectUrl": "https://cvlab-kaist.github.io/Geometric-Action-Model/",
    "venue": "ECCV 2026 Workshop: 3D in the Era of World Models (non-archival)",
    "paperUrl": "https://arxiv.org/abs/2606.17046",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "三维多视角建模",
      "视觉规划与IDM"
    ],
    "architecture": "One Model",
    "predictionParadigm": "IDM",
    "quadrant": "Q2 · One Model × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2606.21088",
    "title": "MV-WAM: Manifold-Aware World Action Model with Value Augmentation",
    "authors": "Chen, Jintao; Jia, Peidong; Wuwu, Qingpo; Liu, Jiaming; Du, Mengfei; Fan, Chun-Kai; Chi, Xiaowei; Chen, Hao; Bai, Chengyu; Qian, Zezhong; Wang, Hao; Cao, Jiajun; Mi, Weishi; Ju, Xiaozhu; Tang, Jian; Zhang, Shanghang",
    "affiliations": "1 State Key Laboratory of Multimedia Information Processing, School of Computer Science, Peking University; 2 Beijing Innovation Center of Humanoid Robotics",
    "contribution": "Recent world action models achieve strong in-domain performance, yet their gains do not extend proportionally to out-of-distribution scenarios. To address this, we propose MV-WAM, a novel end-to-end framework that jointly models visual prediction, action generation, and value estimation designed to effectively leverage video priors during both training and inference for enhanced action generalization.",
    "abstract": "Achieving robust and generalizable manipulation across diverse environments remains a fundamental challenge in embodied robotics. Recent world action models achieve strong in-domain performance, yet their gains do not extend proportionally to out-of-distribution scenarios. We attribute this to a structural mismatch between visual and action modalities, whose intrinsically heterogeneous manifolds cause joint optimization to disproportionately degrade action robustness under distribution shift. To address this, we propose MV-WAM, a novel end-to-end framework that jointly models visual prediction, action generation, and value estimation designed to effectively leverage video priors during both training and inference for enhanced action generalization. Key to this unification is a cross-modality causal mask that hierarchically grounds actions in predicted video frames and value function tokens in both modalities. To further narrow the generalization gap, MV-WAM adopts a manifold-aware optimization scheme that explicitly accounts for the structural heterogeneity across modalities. Finally, MV-WAM introduces a progress-value regulation mechanism that estimates task completion and detects misalignment between predicted frames and generated actions, enabling the policy to autonomously identify execution deviations and recover through value-guided rollback. On the RoboTwin simulation, MV-WAM achieves a 55.7% mean success rate on random scenarios without any randomized action supervision, outperforming the strongest baseline by 29.3%. MV-WAM achieves a 77.5% mean success rate across four real-world tasks of varying difficulty on a dual-arm robot. Our results demonstrate that manifold-aware cross-modal alignment is essential for robust policy generalization, offering a path toward deployable robotic manipulation.",
    "submittedDate": "2026-06-19",
    "primaryCategory": "WAM + RL",
    "secondaryCategories": [],
    "bibtexKey": "MVWAM",
    "arxivUrl": "https://arxiv.org/abs/2606.21088",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2606.21088",
    "pdfUrl": "https://arxiv.org/pdf/2606.21088",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{MVWAM,\n  title={MV-WAM: Manifold-Aware World Action Model with Value Augmentation},\n  author={Chen, Jintao and Jia, Peidong and Wuwu, Qingpo and Liu, Jiaming and Du, Mengfei and Fan, Chun-Kai and Chi, Xiaowei and Chen, Hao and Bai, Chengyu and Qian, Zezhong and Wang, Hao and Cao, Jiajun and Mi, Weishi and Ju, Xiaozhu and Tang, Jian and Zhang, Shanghang},\n  journal={arXiv:2606.21088},\n  year={2026}\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "视觉规划与IDM",
      "泛化与动作对齐",
      "策略后训练与WM-RL"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2606.20781",
    "title": "World Action Models: A Survey",
    "authors": "Shen, Qiuhong; Zhang, Shihua; Liao, Yue; Li, Qi; Tan, Zhenxiong; Wang, Shizun; Yan, Shuicheng; Wang, Xinchao",
    "affiliations": "National University of Singapore",
    "contribution": "World Action Models (WAMs) are embodied predictive-action models that make a forecast of the future available to action. Recent WAMs repurpose large video generation models, and a parallel line relies on language or vision-language backbones without a video-generation core.",
    "abstract": "World Action Models (WAMs) are embodied predictive-action models that make a forecast of the future available to action. Recent WAMs repurpose large video generation models, and a parallel line relies on language or vision-language backbones without a video-generation core. This rapid expansion has blurred the boundary among broad world models, video generation models, action-grounded video world models, Vision-Language-Action policies, and WAMs. This survey gives the field a common account. It first clarifies these boundaries, then organizes existing works through two complementary views. The first view asks what each method is required to generate, spanning rendered futures, latent futures, and video-generation-free action reasoning. The second view decomposes each method by predictive substrate, backbone, action coupling, and deployment regime. This anatomy supports a unified discussion of interactability, causality, persistence, physical plausibility, and generalization, followed by data, evaluation, and open challenges. Across these axes, a consistent design pattern emerges: WAMs are not simply video generators with action heads, but predictive-action methods whose design choices trade representational richness against compute, memory, latency, and action-label cost. The field is moving toward methods that generate less of the future while preserving what control requires. The survey homepage is available at https://world-action-models.github.io/.",
    "submittedDate": "2026-06-18",
    "primaryCategory": "Evaluation / Survey / Theory",
    "secondaryCategories": [
      "Latent / Representation WAM",
      "Memory WAM"
    ],
    "bibtexKey": "WAMSurvey",
    "arxivUrl": "https://arxiv.org/abs/2606.20781",
    "codeUrls": [
      "https://github.com/world-action-models/awesome-world-action-models"
    ],
    "projectUrl": "https://world-action-models.github.io/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2606.20781",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "Related resources",
    "subcategories": [
      "综述与技术资源"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2606.20562",
    "title": "MemoryWAM: Efficient World Action Modeling with Persistent Memory",
    "authors": "Yang, Sizhe; Mu, Juncheng; Wei, Tianming; Lu, Chenhao; Li, Xiaofan; Xu, Linning; Xue, Zhengrong; Yuan, Zhecheng; Lin, Dahua; Pang, Jiangmiao; Xu, Huazhe",
    "affiliations": "1 The Chinese University of Hong Kong; 2 Tsinghua University 3 Zhejiang University 1mm",
    "contribution": "To address this challenge, we introduce MemoryWAM, a world action model with efficient persistent memory. Across long-horizon, memory-dependent manipulation tasks in both simulation and the real world, MemoryWAM outperforms strong vision-language-action (VLA) and WAM baselines while maintaining favorable computational efficiency.",
    "abstract": "Robust robotic manipulation in the real world requires not only an understanding of the current observation, but also memory and dynamics modeling. World action models (WAMs) possess these capabilities by jointly modeling visual foresight and actions conditioned on both current and historical observations, making them a promising paradigm for robotic manipulation. However, existing WAMs face a fundamental trade-off: methods with efficient inference typically condition only on a bounded window of recent observations and therefore struggle in non-Markovian environments, whereas methods that preserve long histories incur time and space costs that grow substantially with sequence length. To address this challenge, we introduce MemoryWAM, a world action model with efficient persistent memory. MemoryWAM uses a hybrid memory design that combines recent frames, event-boundary anchor frames, and compact gist tokens that summarize long-range history. A tailored attention mechanism enables retrieval of both detailed short-term context and compressed long-term context, supporting memory-dependent decision-making with reduced inference latency and GPU memory usage. Across long-horizon, memory-dependent manipulation tasks in both simulation and the real world, MemoryWAM outperforms strong vision-language-action (VLA) and WAM baselines while maintaining favorable computational efficiency.",
    "submittedDate": "2026-06-18",
    "primaryCategory": "Memory WAM",
    "secondaryCategories": [
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "MemoryWAM",
    "arxivUrl": "https://arxiv.org/abs/2606.20562",
    "codeUrls": [],
    "projectUrl": "https://yangsizhe.github.io/MemoryWAM/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2606.20562",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "记忆与长时序",
      "高效推理与实时控制"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2606.20092",
    "title": "EventVLA: Event-Driven Visual Evidence Memory for Long-Horizon Vision-Language-Action Policies",
    "authors": "Ganlin Yang; Zhangzheng Tu; Yuqiang Yang; Sitong Mao; Junyi Dong; Tianxing Chen; Jiaqi Peng; Jing Xiong; Jiafei Cao; Jifeng Dai; Wengang Zhou; Yao Mu; Tai Wang",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2026-06-18",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "yang2026eventvla",
    "arxivUrl": "https://arxiv.org/abs/2606.20092",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2606.20092",
    "pdfUrl": "https://arxiv.org/pdf/2606.20092",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{yang2026eventvla,\n  title={EventVLA: Event-Driven Visual Evidence Memory for Long-Horizon Vision-Language-Action Policies},\n  author={Yang, Ganlin and Tu, Zhangzheng and Yang, Yuqiang and Mao, Sitong and Dong, Junyi and Chen, Tianxing and Peng, Jiaqi and Xiong, Jing and Cao, Jiafei and Dai, Jifeng and Zhou, Wengang and Mu, Yao and Wang, Tai},\n  journal={arXiv:2606.20092},\n  year={2026}\n}",
    "majorCategory": "VLA",
    "subcategories": [
      "未来表征辅助VLA",
      "自回归VLA"
    ],
    "architecture": "One Model",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2606.19531",
    "title": "ImageWAM: Do World Action Models Really Need Video Generation, or Just Image Editing?",
    "authors": "Zhang, Yuyang; Zhang, Wenyao; Qi, Zekun; Zhang, He; Lin, Haitao; Zhang, Jingbo; Mu, Yao; Yang, Xiaokang; Zeng, Wenjun; Jin, Xin",
    "affiliations": "1Shanghai Jiao Tong University, 2Eastern Institute of Technology, 3Tencent Robotics X, 4Tsinghua; University, 5Zhongguancun Academy",
    "contribution": "We propose ImageWAM, a simple WAM framework that repurposes pretrained image editing models for robot action prediction. ImageWAM outperforms standard VLA baselines and matching competitive WAMs without additional policy pretraining across different simulator and real-world experiments.",
    "abstract": "World Action Models (WAMs) commonly rely on video generation to bridge visual world modeling and robot control. However, video-based WAMs face three coupled limitations: dense multi-frame future tokens make inference costly, full video prediction spends capacity on action-irrelevant temporal and appearance details, and long-horizon future imagination may introduce errors that mislead action prediction. These issues raise a simple question: Does world action model really need video generation? We propose ImageWAM, a simple WAM framework that repurposes pretrained image editing models for robot action prediction. In contrast to video generation, image editing provides a better-matched prior: it only needs to model a target-frame transformation, focuses on action-relevant current-to-target visual differences, and grounds task instructions to localized visual changes through edit pretraining. In practice, ImageWAM does not decode the target frame at inference time; instead, it conditions a flow-matching action expert on the KV caches produced by image-editing denoising, using them as a compact world-action context. ImageWAM outperforms standard VLA baselines and matching competitive WAMs without additional policy pretraining across different simulator and real-world experiments. It also reduces FLOPs to 1/6 and latency to 1/4 of video-based WAMs. Attention analysis further shows that editing caches focus on task-relevant change regions, supporting image editing as an effective alternative to video-based world-action modeling.",
    "submittedDate": "2026-06-17",
    "primaryCategory": "Real-Time / Efficient WAM",
    "secondaryCategories": [],
    "bibtexKey": "ImageWAM",
    "arxivUrl": "https://arxiv.org/abs/2606.19531",
    "codeUrls": [
      "https://github.com/yuyangalin/ImageWAM"
    ],
    "projectUrl": "https://zhangwenyao1.github.io/ImageWAM/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2606.19531",
    "pdfUrl": "https://arxiv.org/pdf/2606.19531",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{ImageWAM,\n  title={ImageWAM: Do World Action Models Really Need Video Generation, or Just Image Editing?},\n  author={Zhang, Yuyang and Zhang, Wenyao and Qi, Zekun and Zhang, He and Lin, Haitao and Zhang, Jingbo and Mu, Yao and Yang, Xiaokang and Zeng, Wenjun and Jin, Xin},\n  journal={arXiv:2606.19531},\n  year={2026}\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "视觉规划与IDM",
      "高效推理与实时控制"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2606.17906",
    "title": "WAM-RL: World-Action Model Reinforcement Learning with Reconstruction Rewards and Online Video SFT",
    "authors": "Qian, Zezhong; Chi, Xiaowei; Qi, Yu; Li, Haozhan; Chen, Zhi Yang; Zhang, Shanghang",
    "affiliations": "1 State Key Laboratory of Multimedia Information Processing, School of Computer Science, Peking University; 2 Northeastern University 3 Tsinghua University",
    "contribution": "To address these limitations, we propose WAM-RL, a reinforcement learning framework that enables joint optimization of the world model and the action model through online interaction with the environment. By allowing the two components to co-evolve, our approach enhances fine-grained control and adaptability.",
    "abstract": "Recent World-Action (WA) models demonstrate strong generalization ability and data efficiency, but they typically rely on expert trajectories for training. This reliance limits their ability to acquire fine-grained manipulation skills beyond the demonstration distribution and prevents them from continuously improving through real-world interaction. To address these limitations, we propose WAM-RL, a reinforcement learning framework that enables joint optimization of the world model and the action model through online interaction with the environment. By allowing the two components to co-evolve, our approach enhances fine-grained control and adaptability. Specifically, a WA model consists of a world model and an actor. We design a tailored reinforcement learning method with hierarchical optimization to coordinate their improvement. On the methodological side, we systematically investigate the effects of applying reinforcement learning to the action model, as well as online training of the world model within an RL setting. Our experiments reveal a key insight: optimizing only the actor yields improvements on short-horizon tasks, but fails to provide significant gains on long-horizon tasks. In contrast, jointly optimizing both the world model and the actor is critical for achieving strong performance in long-horizon settings. Our work is the first to introduce reinforcement learning into the World-Action paradigm, and provides insights into how online optimization of both the action head and the world model impacts overall performance.",
    "submittedDate": "2026-06-16",
    "primaryCategory": "WAM + RL",
    "secondaryCategories": [
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "WAMRL",
    "arxivUrl": "https://arxiv.org/abs/2606.17906",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2606.17906",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "策略后训练与WM-RL",
      "视觉规划与IDM"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2606.15869",
    "title": "Metis: A Generalizable and Efficient World-Action Model for Autonomous Driving and Urban Navigation",
    "authors": "Li, Jingyu; Liu, Zhe; Hu, Dongnan; Wu, Junjie; Ma, Zipei; Wu, Wenxiao; Han, Chao; Hao, Zhihui; Liu, Zhikang; Zhan, Kun; Deng, Jiankang; Zhu, Xiatian; Zhang, Li",
    "affiliations": "1 Fudan University; 2 Shanghai Innovation Institute; 3 The University of Hong Kong; 4 Tongji University; 5 Li Auto Inc; 6 Huazhong University of Science and Technology; 7 Imperial College London; 8 University of Surrey",
    "contribution": "To address both issues, we propose Metis, an end-to-end WAM framework that decouples video generation and action prediction. To enhance efficiency, we introduce an asymmetric attention mask that enables joint training of both experts while allowing the action model to bypass explicit video generation during inference.",
    "abstract": "World action models~(WAMs) have shown great promise for autonomous driving and urban navigation. Built upon Vision-Language-Action models or video generation models, existing approaches suffer key limitations: (1) High inference latency due to future observation prediction at test time, and (2) tightly coupled video and action modeling leading to representational mismatch and degraded generalization. To address both issues, we propose Metis, an end-to-end WAM framework that decouples video generation and action prediction. Specifically, Metis employs a Mixture-of-Transformers architecture with dedicated experts for video generation and action prediction, preserving the intrinsic distributional properties of each task. To enhance efficiency, we introduce an asymmetric attention mask that enables joint training of both experts while allowing the action model to bypass explicit video generation during inference. This design ensures training-inference consistency and significantly reduces computational costs without compromising planning performance. Extensive experiments demonstrate state-of-the-art performance on the NAVSIM navhard and navtest benchmarks and the CityWalker navigation benchmark, validating both the generalizability and efficiency across diverse tasks. Real-robot deployments further confirm the practical feasibility of our approach.",
    "submittedDate": "2026-06-14",
    "primaryCategory": "Real-Time / Efficient WAM",
    "secondaryCategories": [
      "Evaluation / Survey / Theory",
      "Navigation / Driving / Domain WAM"
    ],
    "bibtexKey": "Metis",
    "arxivUrl": "https://arxiv.org/abs/2606.15869",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2606.15869",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "自动驾驶",
      "导航",
      "高效推理与实时控制"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2606.15768",
    "title": "LaWAM: Latent World Action Models for Efficient Dynamics-Aware Robot Policies",
    "authors": "Chen, Jialei; Wang, Kai; Chen, Kang; Chen, Shuaihang; Gao, Feng; Tang, Wenhao; Li, Zhiyuan; Liu, Weilin; Yao, Zhuyu; Li, Boxun; Xu, Yuanbo; Yu, Chao",
    "affiliations": "Tsinghua University; Jilin University; Nankai University; Peking University; Harbin Institute of Technology; Zhongguancun Academy",
    "contribution": "We present LaWAM, a Latent World Action Model that exposes predictive dynamics to robot policies through compact latent visual subgoals instead of reconstructed future video. LaWAM achieves state-of-the-art or competitive success rates (SRs) across LIBERO (98.6% SR), RoboTwin (91.22% SR), and real-world manipulation tasks while retaining low-latency inference.",
    "abstract": "Vision-Language-Action models (VLAs) leverage large-scale vision-language pretraining for semantic robot control, but often lack explicit foresight into how robot actions change the scene. World-Action Models (WAMs) address this limitation by conditioning policies on predicted futures, yet existing approaches typically rely on computationally expensive video generation with substantial pixel-level redundancy. We present LaWAM, a Latent World Action Model that exposes predictive dynamics to robot policies through compact latent visual subgoals instead of reconstructed future video. At the core of LaWAM is a latent-action-conditioned Latent World Model (LaWM). We obtain LaWM by training a latent action model in the latent space of a pretrained vision foundation model and repurposing its forward decoder to predict future observation features for scene evolution. LaWAM then conditions action generation on these predicted latent visual subgoals to enable dynamics-aware robot control. LaWAM achieves state-of-the-art or competitive success rates (SRs) across LIBERO (98.6% SR), RoboTwin (91.22% SR), and real-world manipulation tasks while retaining low-latency inference. LaWAM runs in 187 ms per action-chunk prediction and achieves up to 24x lower wall-clock latency than pixel-space WAMs.",
    "submittedDate": "2026-06-14",
    "primaryCategory": "Latent / Representation WAM",
    "secondaryCategories": [
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "LaWAM",
    "arxivUrl": "https://arxiv.org/abs/2606.15768",
    "codeUrls": [
      "https://github.com/RLinf/LaWAM"
    ],
    "projectUrl": "https://rlinf.github.io/LaWAM/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2606.15768",
    "pdfUrl": "https://arxiv.org/pdf/2606.15768",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{LaWAM,\n  title={LaWAM: Latent World Action Models for Efficient Dynamics-Aware Robot Policies},\n  author={Chen, Jialei and Wang, Kai and Chen, Kang and Chen, Shuaihang and Gao, Feng and Tang, Wenhao and Li, Zhiyuan and Liu, Weilin and Yao, Zhuyu and Li, Boxun and Xu, Yuanbo and Yu, Chao},\n  journal={arXiv:2606.15768},\n  year={2026}\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "潜空间预测与JEPA",
      "视觉规划与IDM",
      "高效推理与实时控制"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2606.15631",
    "title": "Retrieve, Don't Retrain: Extending Vision Language Action Models to New Tasks at Test Time",
    "authors": "Park, Jeongeun; Park, Juhan; Kim, Taekyung; Choi, Sungjoon; Han, Dongyoon; Yun, Sangdoo",
    "affiliations": "NAVER AI Lab; Korea University",
    "contribution": "On PushT, we study how retrieval provides a reusable high-level motion prior for cross-embodiment generalization to unseen goal angles, while on RoboTwin 2.0 our method outperforms cross-embodiment baselines on unseen tasks, and we additionally demonstrate the method on a real robot.",
    "abstract": "Extending a vision-language-action (VLA) policy to a new task typically requires task-specific teleoperated demonstrations and per-task fine-tuning, making adaptation costly in both data collection and compute. In this paper, we show that this target-side per-task adaptation cost can be replaced by retrieval. Our retrieval-augmented policy is trained once on paired demonstrations from the target embodiment (query) and a cheaper embodiment (pool, e.g., human-hand video), then frozen. New tasks are added at deployment by appending pool-side demonstrations to a retrieval pool. The frozen policy conditions on retrieved trajectories at every control step, so new tasks are absorbed by indexing data rather than updating parameters. Fine-tuning is needed only to take on a new, unseen embodiment, not for each new task. We show that retrieval improves policies beyond a specific backbone, including standard VLA policies, but its effect is especially pronounced in Cosmos Policy, a video-generation-based world-action model (WAM). In this setting, retrieval supplies coarse task progression, while the WAM's future-image objective provides an additional visual consistency signal that strengthens the retrieval-conditioned actions. On PushT, we study how retrieval provides a reusable high-level motion prior for cross-embodiment generalization to unseen goal angles, while on RoboTwin 2.0 our method outperforms cross-embodiment baselines on unseen tasks, and we additionally demonstrate the method on a real robot.",
    "submittedDate": "2026-06-14",
    "primaryCategory": "Memory WAM",
    "secondaryCategories": [],
    "bibtexKey": "RDTREVLN",
    "arxivUrl": "https://arxiv.org/abs/2606.15631",
    "codeUrls": [
      "https://github.com/jeongeun980906/ReCAP-Cosmos-Policy"
    ],
    "projectUrl": "https://recap-robot.github.io/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2606.15631",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "联合视频动作建模",
      "泛化与动作对齐",
      "记忆与长时序"
    ],
    "architecture": "One Model",
    "predictionParadigm": "联合预测",
    "quadrant": "Q1 · One Model × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2606.20679",
    "title": "MemoryVAM: Integrating Memory into Video Action Model for Robot Manipulation",
    "authors": "Yuxin Jiang; Chang Yu; Yunuo Chen; Xiang Feng; Yin Yang; Nishank Gite; Chenfanfu Jiang",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2026-06-13",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "jiang2026memoryvam",
    "arxivUrl": "https://arxiv.org/abs/2606.20679",
    "codeUrls": [],
    "projectUrl": "https://MemoryVAM.github.io/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2606.20679",
    "pdfUrl": "https://arxiv.org/pdf/2606.20679",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{jiang2026memoryvam,\n  title={MemoryVAM: Integrating Memory into Video Action Model for Robot Manipulation},\n  author={Jiang, Yuxin and Yu, Chang and Chen, Yunuo and Feng, Xiang and Yang, Yin and Gite, Nishank and Jiang, Chenfanfu},\n  journal={arXiv:2606.20679},\n  year={2026}\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "记忆与长时序",
      "视觉规划与IDM"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2606.13674",
    "title": "RepWAM: World Action Modeling with Representation Visual-Action Tokenizers",
    "authors": "Wang, Junke; Zhang, Qihang; Yang, Shuai; Luo, Yiming; Shen, Yujun; Wu, Zuxuan; Jiang, Yu-Gang; Xu, Yinghao",
    "affiliations": "1Institute of Trustworthy Embodied AI, Fudan University; 2Robbyant, Ant Group,3Hongkong University of Science and Technology",
    "contribution": "This work presents RepWAM, a representation-centric world action model (WAM) built on representation visual-action tokenizers.",
    "abstract": "This work presents RepWAM, a representation-centric world action model (WAM) built on representation visual-action tokenizers. Existing WAMs typically inherit reconstruction-oriented video tokenizers from pretrained video generation models. Although these tokenizers preserve visual fidelity, pixel reconstruction alone provides limited guidance for learning instruction-following dynamics that connect future prediction with robot control. To address this, we explore a semantic visual-action latent space for representation-centric world action modeling. Specifically, we train a representation visual-action tokenizer that maps visual inputs into aligned visual and latent action tokens. We then pretrain our WAM to jointly model future visual states and the latent actions that connect them under language instructions, followed by adaptation to real robot trajectories for closed-loop manipulation. Experiments on real-world manipulation tasks and simulation benchmarks show that RepWAM delivers strong performance across diverse manipulation settings, while ablations highlight the value of semantic visual-action tokenization over reconstruction-oriented alternatives. These results establish representation visual-action tokenization as a promising foundation for world action models and a step toward generalist robot policies. Code and weights will be available at https://github.com/wdrink/RepWAM.",
    "submittedDate": "2026-06-13",
    "primaryCategory": "Latent / Representation WAM",
    "secondaryCategories": [],
    "bibtexKey": "RepWAM",
    "arxivUrl": "https://arxiv.org/abs/2606.13674",
    "codeUrls": [],
    "projectUrl": "https://wdrink.github.io/RepWAM/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2606.13674",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "潜空间预测与JEPA",
      "联合视频动作建模",
      "泛化与动作对齐"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "联合预测",
    "quadrant": "Q3 · Dual-system × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2606.04907",
    "title": "WAM-Nav: Asymmetric Latent World-Action Modeling for Unified Visual Navigation",
    "authors": "Yang, Ning; Huang, Yan; Peng, Kaiwen; He, Ziheng; Wang, Kai; Miao, Cui; Lyu, Kailin; Li, Guo; Wang, Xiaofeng; Zhu, Zheng; Liu, Jing; Liu, Nianfeng",
    "affiliations": "1 Nanjing University 2 Institute of Automation, Chinese Academy of Sciences; 3 University of Chinese Academy of Sciences 4 FiveAges; 5 National University of Defense Technology 6 Tsinghua University 7 GigaAI",
    "contribution": "To address these limitations, we propose WAM-Nav, a Latent World-Action Model for embodied visual navigation that jointly learns action generation and latent visual foresight, enabling more robust and foresighted navigation decisions without compromising inference efficiency. To further encourage smooth and consistent trajectory generation, we introduce a dual-stream contextual conditioning mechanism that integrates episode-level ego-motion history with sequential visual observations.",
    "abstract": "Visual navigation requires generating smooth and collision-free trajectories under complex geometric and physical constraints. Existing reactive policies that directly map observations to actions lack anticipatory reasoning, limiting their ability to proactively avoid obstacles. While visual imagination offers predictive foresight, conventional modular approaches separate scene prediction from policy learning, often leading to error accumulation and inefficient inference. To address these limitations, we propose WAM-Nav, a Latent World-Action Model for embodied visual navigation that jointly learns action generation and latent visual foresight, enabling more robust and foresighted navigation decisions without compromising inference efficiency. Specifically, WAM-Nav utilizes a shared Diffusion Transformer for asymmetric joint diffusion to concurrently generate long-horizon actions and short-horizon visual foresight, reducing the inference latency and visual error accumulation inherent in multi-step autoregressive rollouts. To further encourage smooth and consistent trajectory generation, we introduce a dual-stream contextual conditioning mechanism that integrates episode-level ego-motion history with sequential visual observations. Combined with a unified goal alignment module that preserves balanced representations across goal types, WAM-Nav naturally supports Image-Goal, Point-Goal, and No-Goal exploration within a single policy. Extensive experiments on the challenging ClutterScenes and InternScenes benchmarks demonstrate strong generalization of WAM-Nav, particularly on Image-Goal and Point-Goal navigation, where it improves success rates by 15.7% and 3.3%, respectively. Real-world deployment further validates effective zero-shot sim-to-real transfer, achieving an average 85% task success rate across diverse indoor and outdoor environments.",
    "submittedDate": "2026-06-13",
    "primaryCategory": "Latent / Representation WAM",
    "secondaryCategories": [
      "3D/4D WAM",
      "Navigation / Driving / Domain WAM",
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "WAMNav",
    "arxivUrl": "https://arxiv.org/abs/2606.04907",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2606.04907",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "导航",
      "潜空间预测与JEPA",
      "联合视频动作建模"
    ],
    "architecture": "One Model",
    "predictionParadigm": "联合预测",
    "quadrant": "Q1 · One Model × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2606.13515",
    "title": "MaskWAM: Unifying Mask Prompting and Prediction for World-Action Models",
    "authors": "Yu, Hanyang; Lin, Haitao; Zhang, Jingbo; Zhang, Wenyao; Gu, Chenghao; Li, Heng; Tan, Ping",
    "affiliations": "1The Hong Kong University of Science and Technology, 2Tencent Robotics X, 3Tsinghua University; ‡Work done during an internship at Tencent Robotics X, †Corresponding author",
    "contribution": "To overcome these limitations, we introduce MaskWAM, an object-centric world-action model. Evaluations on LIBERO, RoboTwin, and real-world tasks demonstrate that MaskWAM significantly outperforms baselines in both language-clear and language-ambiguous tasks.",
    "abstract": "World Action Models (WAMs) present a promising paradigm for robotic control via video prediction. However, current WAMs suffer from fundamental spatial bottlenecks: standard text inputs introduce referential ambiguity in cluttered scenes, while unstructured RGB predictions lack semantic grounding and remain biased by task-irrelevant backgrounds. To overcome these limitations, we introduce MaskWAM, an object-centric world-action model. By jointly integrating masks as both explicit inputs and predictions via a unified Mixture of Transformers (MoT), MaskWAM unlocks robust policy generalization. This design provides two key benefits: (1) predicting future masks yields object-centric semantic supervision that suppresses visual noise, significantly enhancing even standard text-conditioned WAMs; and (2) coupling this predictive supervision with first-frame visual prompts, such as target object masks, establishes a precise spatial anchor that substantially reduces language ambiguity. Crucially, as WAMs are inherently vision-driven architectures, direct mask conditioning yields substantially stronger guidance than text alone, establishing a precise and robust paradigm for manipulating unseen objects. Evaluations on LIBERO, RoboTwin, and real-world tasks demonstrate that MaskWAM significantly outperforms baselines in both language-clear and language-ambiguous tasks.",
    "submittedDate": "2026-06-11",
    "primaryCategory": "General WAM",
    "secondaryCategories": [],
    "bibtexKey": "MaskWAM",
    "arxivUrl": "https://arxiv.org/abs/2606.13515",
    "codeUrls": [],
    "projectUrl": "https://hanyangyu1021.github.io/maskwam.github.io/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2606.13515",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "视觉规划与IDM",
      "泛化与动作对齐"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2606.13494",
    "title": "NavWAM: A Navigation World Action Model for Goal-Conditioned Visual Navigation",
    "authors": "Azuma, Daichi; Miyanishi, Taiki; Sakamoto, Koya; Kurita, Shuhei; Zhu, Yaonan; Khrapchenkov, Petr; Kawanabe, Motoaki; Iwasawa, Yusuke; Matsuo, Yutaka",
    "affiliations": "The University of Tokyo, Japan; 1 The University of Tokyo 2 National Institute of",
    "contribution": "We propose Navigation World Action Model (NavWAM), a diffusion-transformer policy that turns navigation world-model prediction into executable action by representing future observations, goal-progress values, and action chunks in a shared latent sequence. We build NavWAM through simulation pretraining and real-robot adaptation, and evaluate it on image-goal navigation against planning-based world models and a representative direct navigation policy.",
    "abstract": "Goal-conditioned visual navigation requires a robot to act under partial observability by anticipating how its motion will change the future egocentric view and whether that change brings it closer to the goal. Navigation world models provide such visual foresight, but they remain prediction modules that require an external planner to convert predicted futures into closed-loop control. We propose Navigation World Action Model (NavWAM), a diffusion-transformer policy that turns navigation world-model prediction into executable action by representing future observations, goal-progress values, and action chunks in a shared latent sequence. By learning future prediction jointly with the action and value targets that determine closed-loop behavior, NavWAM makes visual foresight directly usable for robot control. We build NavWAM through simulation pretraining and real-robot adaptation, and evaluate it on image-goal navigation against planning-based world models and a representative direct navigation policy. Across offline benchmarks and closed-loop real-robot deployment, NavWAM improves over planning-based world-model baselines in our evaluations while using the default policy mode without CEM-style action search. Project page: https://dachii-azm.github.io/navwam/",
    "submittedDate": "2026-06-11",
    "primaryCategory": "Navigation / Driving / Domain WAM",
    "secondaryCategories": [
      "Latent / Representation WAM"
    ],
    "bibtexKey": "NavWAM",
    "arxivUrl": "https://arxiv.org/abs/2606.13494",
    "codeUrls": [],
    "projectUrl": "https://dachii-azm.github.io/navwam/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2606.13494",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "导航",
      "联合视频动作建模"
    ],
    "architecture": "One Model",
    "predictionParadigm": "联合预测",
    "quadrant": "Q1 · One Model × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2606.12987",
    "title": "Diffusion Transformer World-Action Model for AV Scene Prediction",
    "authors": "Sharifullin, Ruslan; Jiang, Benjamin; Chew, Kai Xi",
    "affiliations": "Stanford University",
    "contribution": "Action-conditioned world models let an autonomous vehicle predict future camera scenes from its own planned controls, enabling planning and simulation without real-world rollouts, but at compact, trainable scale the futures are ambiguous and the field's standard distortion metrics actively mislead: they reward a blurry regression mean over a realistic prediction. We confront this with a compact latent world model that, given the present front-camera latent and a sequence of ego-actions, predicts future scene latents a frozen decoder renders to 256×256256 \\times 256 frames up to 8 seconds ahead, evaluated on 150 held-out nuScenes scenes.",
    "abstract": "Action-conditioned world models let an autonomous vehicle predict future camera scenes from its own planned controls, enabling planning and simulation without real-world rollouts, but at compact, trainable scale the futures are ambiguous and the field's standard distortion metrics actively mislead: they reward a blurry regression mean over a realistic prediction. We confront this with a compact latent world model that, given the present front-camera latent and a sequence of ego-actions, predicts future scene latents a frozen decoder renders to 256×256256 \\times 256 frames up to 8 seconds ahead, evaluated on 150 held-out nuScenes scenes. We first benchmark where to predict: across six frozen encoders spanning four representation families, V-JEPA2 with temporal context reduces steering RMSE by 40% over the best single-frame encoder. We then train a latent Diffusion Transformer (DiT) and, through a controlled diagnosis, identify the four ingredients it needs: spatial tokens, the x0x_0 objective, residual anchoring, and sampling matched to target uncertainty. In a Stable-Diffusion-VAE encode-predict-decode pipeline we expose the central tension: distortion metrics (cosine similarity, SSIM) favor the blurry mean, masking that the diffusion model is far closer to the real frame distribution. Inception-based FID and KID reveal a clean perception-distortion frontier: diffusion attains KID 0.078 versus 0.375 for regression (4.8×4.8\\times better), and a deployable train-derived calibration makes this practical without test-time ground truth. The model is genuinely action-controllable (steering drives scene displacement, Spearman ρ=0.81ρ= 0.81, vs −0.18-0.18 for regression). We trace limited single-pass motion to a shared-present anchor and engineer a compact 1.7M-parameter \"jump\" model that recovers full ground-truth motion magnitude (1.02×1.02\\times GT), where single-pass models capture less than half.",
    "submittedDate": "2026-06-11",
    "primaryCategory": "Latent / Representation WAM",
    "secondaryCategories": [
      "Evaluation / Survey / Theory",
      "Navigation / Driving / Domain WAM",
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "DiT",
    "arxivUrl": "https://arxiv.org/abs/2606.12987",
    "codeUrls": [
      "https://github.com/dlcv-team/latent-world-models-av"
    ],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2606.12987",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "神经世界模拟器",
      "自动驾驶"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2606.08555",
    "title": "FAWAM: Force-Aware World Action Models for Closed-Loop Contact-Rich Manipulation",
    "authors": "He, Haotian; Yan, Zeyu; Liu, Qipeng; Guo, Ning; Lian, Wenzhao",
    "affiliations": "1 School of Mathematical Sciences, Peking University; 2 School of Artificial Intelligence, Shanghai Jiao Tong University",
    "contribution": "In this paper, we propose FAWAM, a force-aware world action model that incorporates force information at three levels: perception, prediction, and closed-loop execution.",
    "abstract": "Force signals provide critical interaction cues for contact-rich robotic manipulation. However, existing methods mostly use force as an additional observation modality, without fully exploiting its role in modeling future interaction dynamics or guiding execution-time feedback correction. In this paper, we propose FAWAM, a force-aware world action model that incorporates force information at three levels: perception, prediction, and closed-loop execution. FAWAM first encodes historical 6-axis force/torque signals to modulate action generation, then jointly predicts future actions and end-effector wrenches to explicitly model contact evolution. It further introduces a residual correction module that uses the predicted wrench trajectory as an execution-time reference to refine actions online based on real-time force feedback. Real-world experiments across multiple contact-rich tasks show that FAWAM improves the average success rate by 36.25% over vision-only baselines and 21.25% over existing force-aware baselines, demonstrating the effectiveness of our force-aware framework for robust contact-rich manipulation.",
    "submittedDate": "2026-06-11",
    "primaryCategory": "Multimodal / Tactile WAM",
    "secondaryCategories": [
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "FAWAM",
    "arxivUrl": "https://arxiv.org/abs/2606.08555",
    "codeUrls": [
      "https://github.com/HaotianHehaha/FAWAM"
    ],
    "projectUrl": "https://fawam.github.io/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2606.08555",
    "pdfUrl": "https://arxiv.org/pdf/2606.08555",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{FAWAM,\n  title={FAWAM: Force-Aware World Action Models for Closed-Loop Contact-Rich Manipulation},\n  author={He, Haotian and Yan, Zeyu and Liu, Qipeng and Guo, Ning and Lian, Wenzhao},\n  journal={arXiv:2606.08555},\n  year={2026}\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "多模态触觉音频",
      "联合视频动作建模"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "联合预测",
    "quadrant": "Q3 · Dual-system × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2606.12690",
    "title": "EWAM: An Enhanced World Action Model for Closed-Loop Online Adaptation in Embodied Intelligence",
    "authors": "Zhou, Xin; Miao, Cong",
    "affiliations": "Astronex Robotics; Nanjing University of Information Science and Technology",
    "contribution": "In this paper, we propose the Enhanced World Action Model (EWAM), a closed-loop online adaptation architecture built upon a pretrained and fully frozen Cosmos3 backbone network.",
    "abstract": "In this paper, we propose the Enhanced World Action Model (EWAM), a closed-loop online adaptation architecture built upon a pretrained and fully frozen Cosmos3 backbone network. Evaluated entirely under a zero-shot task protocol, EWAM is centrally focused on reducing the amount of additional deployment data required to adapt to new task layouts. Notably, no extra task-specific demonstration sets were introduced in any of the evaluations, and no fine-tuning was performed on the backbone network. Its performance gains stem entirely from an inference-time co-reasoning mechanism composed of four inserted lightweight neural layers: the Neural Experience Memory Layer located in the intermediate layers of the Diffusion Transformer (DiT) provides task-relevant execution context; the Neural Anomaly Detection Layer after the state prediction head monitors the divergence between predicted and actual states in real time; the Neural Policy Routing Layer dynamically selects direct execution, conservative replanning, or rollback recovery based on the anomaly severity; and the Neural Action Correction Layer refines the generated action chunks using execution diagnostics. Unlike naive feature fusion, the memory, anomaly detection, and correction modules are deeply integrated into the Cosmos3 forward path in a differentiable manner, with only the final routing decision being a discrete supervised one.",
    "submittedDate": "2026-06-10",
    "primaryCategory": "Memory WAM",
    "secondaryCategories": [
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "EWAM",
    "arxivUrl": "https://arxiv.org/abs/2606.12690",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2606.12690",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "记忆与长时序",
      "策略后训练与WM-RL"
    ],
    "architecture": "One Model",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2606.12403",
    "title": "World Pilot: Steering Vision-Language-Action Models with World-Action Priors",
    "authors": "Lin, Zefu; Cui, Rongxu; Xu, Junjia; Jin, Xiaojuan; Li, Wenling; Fan, Lue; Zhang, Zhaoxiang",
    "affiliations": "1 Institute of Automation, Chinese Academy of Sciences (CASIA); 2 Nanjing University; 3 Beihang University",
    "contribution": "We present World Pilot, a VLA framework that augments the policy with priors from a World-Action Model (WAM), routed into the decision chain through two complementary pathways. World Pilot attains a state-of-the-art Total success rate of 84.7% on the LIBERO-Plus zero-shot OOD benchmark and the highest success rate on every real-robot setting across four manipulation tasks, with the largest margins under shifts in viewpoint, geometry, deformable state, and pose.",
    "abstract": "Vision-Language-Action (VLA) models inherit semantic grounding from large-scale pretraining and perform competently across in-distribution manipulation tasks. This grounding, however, is built on static image-text pairs, whereas manipulation is a continuous, contact-rich process whose dynamics this pretraining cannot capture. We present World Pilot, a VLA framework that augments the policy with priors from a World-Action Model (WAM), routed into the decision chain through two complementary pathways. Latent Steering conditions the perception layer on a scene-evolution latent, and Action Steering supplies an anticipated trajectory as a motion prior to the action generator. Together the two priors equip the VLA with an anticipated view of the scene and a trajectory-level motion hint alongside its semantic conditioning, and the scene-evolution prior remains effective even when supplied by a video-pretrained world model that has not been action-post-trained. World Pilot attains a state-of-the-art Total success rate of 84.7% on the LIBERO-Plus zero-shot OOD benchmark and the highest success rate on every real-robot setting across four manipulation tasks, with the largest margins under shifts in viewpoint, geometry, deformable state, and pose. Project Website: https://world-pilot.github.io/",
    "submittedDate": "2026-06-10",
    "primaryCategory": "3D/4D WAM",
    "secondaryCategories": [
      "Evaluation / Survey / Theory",
      "Latent / Representation WAM"
    ],
    "bibtexKey": "WorldPilot",
    "arxivUrl": "https://arxiv.org/abs/2606.12403",
    "codeUrls": [
      "https://github.com/ZefuLin/WorldPilot"
    ],
    "projectUrl": "https://world-pilot.github.io/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2606.12403",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "视觉规划与IDM",
      "潜空间预测与JEPA",
      "泛化与动作对齐"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2606.12217",
    "title": "Making Foresight Actionable: Repurposing Representation Alignment in World Action Models",
    "authors": "Qiu, Lu; Li, Yizhuo; Chen, Yi; Ge, Yuying; Ge, Yixiao; Liu, Xihui",
    "affiliations": "1 The University of Hong Kong 2 XPENG Robotics; Project Page:",
    "contribution": "In this paper, we propose AGRA, an Action-Grounded Representation Alignment objective that regularizes the world-action interface by aligning intermediate video diffusion features with spatially coherent semantic representations from a foundation visual encoder.",
    "abstract": "World Action Models (WAMs) offer a promising route for robot manipulation by using video generation models to model future scene evolution before producing control actions. However, our empirical observations reveal a phenomenon: generating plausible visual futures does not always guarantee the extraction of accurate actions. To diagnose this failure, we conduct action-head attention analysis and causal interventions. We find that the action decoder fails to focus on task-relevant interaction regions and remains sensitive to perturbations in task-irrelevant areas. This reveals a representation mismatch: hidden states optimized for visual reconstruction are not inherently organized in a form useful for low-level action control. In this paper, we propose AGRA, an Action-Grounded Representation Alignment objective that regularizes the world-action interface by aligning intermediate video diffusion features with spatially coherent semantic representations from a foundation visual encoder. We evaluate AGRA on real-world manipulation tasks. Experiments show that AGRA makes world model representations more action-grounded: by focusing the action decoder on the correct interaction regions, it improves object localization accuracy and affordance understanding, and makes the policy more robust to perturbations in task-irrelevant regions. As a result, AGRA consistently improves both in-distribution performance and out-of-distribution generalization over the baseline world action model.",
    "submittedDate": "2026-06-10",
    "primaryCategory": "General WAM",
    "secondaryCategories": [],
    "bibtexKey": "GRA",
    "arxivUrl": "https://arxiv.org/abs/2606.12217",
    "codeUrls": [],
    "projectUrl": "https://xpeng-robotics.github.io/agra/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2606.12217",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "视觉规划与IDM",
      "泛化与动作对齐"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2606.10040",
    "title": "Efficient-WAM: A 1B-Parameter World-Action Model with Low-Cost Future Imagination",
    "authors": "Li, Jiajun; Guo, Tiecheng; Ye, Yifan; Zhang, Rongyu; Chi, Xiaowei; Sun, Qianpu; Li, Ying; Lou, Yunfan; Huang, Yan; Lu, Zhihe; Guo, Meng; Zhang, Shanghang",
    "affiliations": "The University of Hong Kong; Peking University; Muka Robotics; Institute of Automation, Chinese Academy of Sciences; Nanjing University",
    "contribution": "We introduce Efficient-WAM, a World-Action Model that reduces the cost of future imagination while preserving its control benefit.",
    "abstract": "World-Action Models (WAMs) have emerged as a promising paradigm for embodied control by coupling future visual prediction with action generation. However, most existing WAMs rely on photorealistic future prediction, which incurs high inference latency and makes real-time robot deployment difficult. This motivates a more efficient WAM design that preserves the control benefits of future visual prediction while reducing its inference cost. We introduce Efficient-WAM, a World-Action Model that reduces the cost of future imagination while preserving its control benefit. Efficient-WAM improves inference efficiency via a compact video expert transferred from WAN-2.2-5B, token-sparse video latents, and asymmetric video-action denoising that allocates fewer sampling steps to video than to actions. Instead of optimizing the future branch for visual fidelity, Efficient-WAM treats future video prediction as a compact guidance signal for action generation. Comprehensive experiments on RoboTwin 2.0 and real-world manipulation tasks show that Efficient-WAM maintains strong action performance despite visibly coarse future predictions. While maintaining competitive control capabilities, our 1B-parameter model can reduce per-chunk latency to around 100 ms during physical deployment, achieving a 30x speedup over existing WAMs.",
    "submittedDate": "2026-06-10",
    "primaryCategory": "Real-Time / Efficient WAM",
    "secondaryCategories": [],
    "bibtexKey": "EfficientWAM",
    "arxivUrl": "https://arxiv.org/abs/2606.10040",
    "codeUrls": [
      "https://github.com/jiajun613/Efficient-WAM"
    ],
    "projectUrl": "https://efficientwam.github.io/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2606.10040",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "视觉规划与IDM",
      "高效推理与实时控制"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2606.11187",
    "title": "Next Forcing: Causal World Modeling with Multi-Chunk Prediction",
    "authors": "Xu, Gangwei; Zhang, Qihang; Zhou, Jiaming; Zhu, Xing; Shen, Yujun; Yang, Xin; Xu, Yinghao",
    "affiliations": "Not identified",
    "contribution": "In this paper, we present Next Forcing, a multi-chunk prediction (MCP) framework for causal world modeling that enables faster training, higher accuracy, and accelerated inference. During training, the MCP modules significantly accelerate convergence and improve converged accuracy, especially at high frame rates: at 50 fps, Next Forcing achieves a 93.1% relative improvement over LingBot-VA at 5k training steps and 2.3x faster convergence, and establishes new state-of-the-art results on the RoboTwin benchmark (94.1/93.5% on Clean/Random).",
    "abstract": "Autoregressive video generation has emerged as a powerful paradigm for World Action Models (WAMs). However, existing approaches suffer from slow training convergence and limited converged accuracy, particularly at high frame rates, as the training supervision is confined to the current chunk without explicit signals about future dynamics; they also suffer from slow inference due to iterative video denoising. In this paper, we present Next Forcing, a multi-chunk prediction (MCP) framework for causal world modeling that enables faster training, higher accuracy, and accelerated inference. Inspired by multi-token prediction in large language models, Next Forcing introduces an MCP training objective that augments the main model with lightweight auxiliary MCP modules to simultaneously denoise video chunks at multiple future temporal horizons (next1^1, next2^2, next3^3 chunks). These MCP modules form a causal chain across prediction depths, where intermediate features fused from multiple layers of the main model are leveraged to predict future dynamics, allowing near-future predictions to inform farther-future ones and providing dense multi-scale temporal supervision back to the main model. During training, the MCP modules significantly accelerate convergence and improve converged accuracy, especially at high frame rates: at 50 fps, Next Forcing achieves a 93.1% relative improvement over LingBot-VA at 5k training steps and 2.3x faster convergence, and establishes new state-of-the-art results on the RoboTwin benchmark (94.1/93.5% on Clean/Random). At inference, the MCP modules can be retained to predict the next video chunk in parallel with the current one, achieving 2x inference acceleration. Next Forcing also demonstrates significant improvements on PhyWorld, a benchmark evaluating adherence to physical laws in video generation, and over 50% FVD reduction on general video pretraining.",
    "submittedDate": "2026-06-09",
    "primaryCategory": "Real-Time / Efficient WAM",
    "secondaryCategories": [
      "Evaluation / Survey / Theory"
    ],
    "bibtexKey": "MCP",
    "arxivUrl": "https://arxiv.org/abs/2606.11187",
    "codeUrls": [
      "https://github.com/gangweix/next-forcing"
    ],
    "projectUrl": "https://gangweix.github.io/next-forcing/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2606.11187",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "视觉规划与IDM",
      "记忆与长时序",
      "高效推理与实时控制"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2606.10363",
    "title": "HiMem-WAM: Hierarchical Memory-Gated World Action Models for Robotic Manipulation",
    "authors": "Sun, Xiaoquan; Zhang, Ruijian; Cao, Chen; Sun, Yihan; Chen, Jiahui; Xu, Zetian; Chen, Bo; Chen, Haijier; Yang, Zhen; Zhu, Jiarun; Hong, Yijun; Xu, JingZhe; Pang, Jingrui; Yuan, Mingqi; Chen, Jiayu",
    "affiliations": "The University of Hong Kong; Huazhong University of Science and Technology; Tsinghua University; Wuhan University; Southern University of Science and Technology",
    "contribution": "To address this, we present HiMem-WAM, a Hierarchical Memory-Gated WAM that integrates motion-centric latent actions, high-level skill latents, and boundary-triggered memory updates. Specifically, we develop a hierarchical latent action framework that jointly learns low-level motion and high-level skill latents, providing structured temporal abstraction.",
    "abstract": "World Action Models (WAMs) have emerged as a new powerful paradigm for embodied intelligence, learning action-relevant visual dynamics that significantly enhance generalization and robustness. However, existing WAMs still struggle with task-relevant memory in long-horizon robotic manipulation. To address this, we present HiMem-WAM, a Hierarchical Memory-Gated WAM that integrates motion-centric latent actions, high-level skill latents, and boundary-triggered memory updates. Specifically, we develop a hierarchical latent action framework that jointly learns low-level motion and high-level skill latents, providing structured temporal abstraction. Meanwhile, a boundary-aware memory gate writes compact task states at predicted skill transitions, enabling causal inference without test-time generation of future video or optical flow estimation. Evaluated on LIBERO, LIBERO-PLUS, RMBench and real-world tasks, HiMem-WAM shows that hierarchical latents improve robustness under deployment perturbations, and the memory module substantially benefits memory-dependent long-horizon manipulation.",
    "submittedDate": "2026-06-08",
    "primaryCategory": "Memory WAM",
    "secondaryCategories": [
      "Latent / Representation WAM",
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "HiMemWAM",
    "arxivUrl": "https://arxiv.org/abs/2606.10363",
    "codeUrls": [
      "https://github.com/Agentic-Intelligence-Lab/HiMem-WAM"
    ],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2606.10363",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "VLA",
    "subcategories": [
      "分层与双系统VLA",
      "潜动作预训练"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2606.09811",
    "title": "AHA-WAM:Asynchronous Horizon-Adaptive World-Action Modeling with Observation-Guided Context Routing",
    "authors": "Cai, Jisong; Ling, Long; Chu, Shiwei; Liu, Zhongshan; Kang, Jiayue; Liang, Zhixuan; Xu, Wenjie; Mao, Yinan; Zhang, Weinan; Yang, Xiaokang; Ying, Ru; Zheng, Ran; Mu, Yao",
    "affiliations": "1 Shanghai Jiao Tong University 2 Shanghai AI Laboratory; 3 Baidu AI Cloud 4 The University of Hong Kong",
    "contribution": "Therefore, we propose AHA-WAM, an Asynchronous Horizon-Adaptive World-Action Model built on a dual Diffusion Transformer (DiT) architecture that reorganizes world-action modeling around this temporal asymmetry. To support asynchronous execution, we introduce horizon-adaptive offset training and Observation-Guided Video-Context Routing (OVCR), which together let the action expert exploit long-horizon world context while remaining responsive to real-time execution state without rerunning the video DiT.",
    "abstract": "World-action models have emerged as a promising paradigm for robot manipulation, jointly modeling visual scene dynamics and actions to inject physical priors into policy learning. However, existing world-action models couple world prediction and action execution at the same temporal resolution, forcing the world branch to model near-term frame variations that are redundant and weakly informative. We posit that strictly binding world prediction and action execution to the same temporal rhythm may underutilize the potential of the video branch for embodied control. Therefore, we propose AHA-WAM, an Asynchronous Horizon-Adaptive World-Action Model built on a dual Diffusion Transformer (DiT) architecture that reorganizes world-action modeling around this temporal asymmetry. AHA-WAM instantiates the video DiT as a low-frequency world planner that maintains rolling key-value memory over past observations and exposes reusable layerwise latent context encoding long-horizon scene evolution, while a high-frequency action DiT executes short action chunks in closed loop by querying this context through layerwise joint attention. To support asynchronous execution, we introduce horizon-adaptive offset training and Observation-Guided Video-Context Routing (OVCR), which together let the action expert exploit long-horizon world context while remaining responsive to real-time execution state without rerunning the video DiT. Experiments on RoboTwin and real-world manipulation tasks show that AHA-WAM achieves state-of-the-art performance without any robot-data pretraining, attaining 92.80% average success on RoboTwin and 78.3% success across 4 real-world tasks, while reaching 24.17 Hz closed-loop control with a 4.59x speedup over Fast-WAM.",
    "submittedDate": "2026-06-08",
    "primaryCategory": "Real-Time / Efficient WAM",
    "secondaryCategories": [
      "Latent / Representation WAM",
      "Memory WAM"
    ],
    "bibtexKey": "AHAWAM",
    "arxivUrl": "https://arxiv.org/abs/2606.09811",
    "codeUrls": [
      "https://github.com/serene-sivy/AHA-WAM"
    ],
    "projectUrl": "https://serene-sivy.github.io/aha-wam/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2606.09811",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "记忆与长时序",
      "高效推理与实时控制",
      "联合视频动作建模"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "联合预测",
    "quadrant": "Q3 · Dual-system × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2606.09215",
    "title": "MotionWAM: Towards Foundation World Action Models for Real-Time Humanoid Loco-Manipulation",
    "authors": "Zheng, Jia; Ma, Teli; Fan, Yudong; Wang, Zifan; Yang, Shuo; Liang, Junwei",
    "affiliations": "1 Mondo Robotics 2 HKUST (GZ) 3 HKUST",
    "contribution": "We present MotionWAM, a real-time WAM that drives autonomous humanoid loco-manipulation from a single egocentric camera by conditioning the policy on the intermediate denoising features of a video world model. On nine real-world Unitree G1 tasks, MotionWAM runs in real time, substantially outperforms Vision-Language-Action (VLA) baselines fine-tuned on the same demonstrations by over 30% in overall success rate, and executes task-driven foot interaction that decoupled upper-lower policies cannot reach.",
    "abstract": "World Action Models (WAMs) couple a video dynamics prior to the policy and have shown encouraging results on tabletop manipulation, but iterative denoising over high-dimensional video-action latents leaves them too slow for real-time humanoid loco-manipulation. The problem is compounded by the dominant hierarchical paradigm, in which a high-level manipulation policy controls only the upper body while a low-level controller tracks coarse base commands -- placing upper and lower body in inconsistent action spaces and reducing the legs to balance-preserving locomotion. We present MotionWAM, a real-time WAM that drives autonomous humanoid loco-manipulation from a single egocentric camera by conditioning the policy on the intermediate denoising features of a video world model. MotionWAM replaces the upper-lower split with a unified motion latent and predicts whole-body motion tokens that jointly cover locomotion, torso motion, height regulation, foot interaction, and hand manipulation in a single action space. A three-stage learning framework progressively adapts the video world model to egocentric visual dynamics and to the target humanoid embodiment. On nine real-world Unitree G1 tasks, MotionWAM runs in real time, substantially outperforms Vision-Language-Action (VLA) baselines fine-tuned on the same demonstrations by over 30% in overall success rate, and executes task-driven foot interaction that decoupled upper-lower policies cannot reach. Our results suggest that video-pretrained WAMs can be lifted from tabletop manipulation to coordinated, human-like whole-body humanoid control.",
    "submittedDate": "2026-06-08",
    "primaryCategory": "Real-Time / Efficient WAM",
    "secondaryCategories": [
      "Latent / Representation WAM",
      "Navigation / Driving / Domain WAM"
    ],
    "bibtexKey": "MotionWAM",
    "arxivUrl": "https://arxiv.org/abs/2606.09215",
    "codeUrls": [],
    "projectUrl": "https://dit4dit.github.io/MotionWAM/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2606.09215",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "导航",
      "视觉规划与IDM",
      "泛化与动作对齐"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2606.08962",
    "title": "C3^3ache: Accelerating World Action Models with Cross Inference Chunk Cache",
    "authors": "Zhao, Weisen; Nguyen, Lam; Lu, Zhicong; Shang, Yuzhang",
    "affiliations": "George Mason University; University of Central Florida",
    "contribution": "We introduce C3^3ache, a training-free method that caches and reuses these residuals across inference chunks at the same denoising step. Experiments on benchmarks with a Fast-WAM backbone show that C3^3ache achieves up to a 2.5×2.5\\times speedup in total wall-clock inference time, with negligible degradation in task success rate.",
    "abstract": "World Action Models (WAMs) generalize better than standard Vision-Language-Action (VLA) policies to novel motions and environments, because a video-modeling objective lets them learn from abundant unlabeled video rather than scarce labeled robot demonstrations. This generalization is computationally expensive. To complete a task, a WAM runs over multiple inference chunks, and each chunk requires a costly denoising process. Existing acceleration methods reduce this cost by caching and reusing computation within a single chunk's denoising trajectory. Our empirical analysis reveals a substantial source of redundancy they overlook: redundancy across chunks. When a robot executes a smooth behavior, the residuals computed at a given denoising step are strongly correlated from one chunk to the next. We introduce C3^3ache, a training-free method that caches and reuses these residuals across inference chunks at the same denoising step. Experiments on benchmarks with a Fast-WAM backbone show that C3^3ache achieves up to a 2.5×2.5\\times speedup in total wall-clock inference time, with negligible degradation in task success rate.",
    "submittedDate": "2026-06-07",
    "primaryCategory": "Real-Time / Efficient WAM",
    "secondaryCategories": [],
    "bibtexKey": "C33ache",
    "arxivUrl": "https://arxiv.org/abs/2606.08962",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2606.08962",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "高效推理与实时控制"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2606.08737",
    "title": "Dream-Tac: A Unified Tactile World Action Model for Contact-Rich Robot Manipulation",
    "authors": "Lou, Yunfan; Ye, Yifan; Fu, Yankai; Cen, Jun; Chi, Xiaowei; Lyu, Yaoxu; Jia, Peidong; Han, Sirui; Lu, Zhihe; Zhang, Shanghang",
    "affiliations": "Peking University; Beijing; China; The Hong Kong University of Science and Technology; Hong Kong; Nanjing University; Nanjing; State Key Laboratory of Multimedia Information Processing, School of Computer Science, Peking University",
    "contribution": "In this paper, we propose Dream-Tac, a unified Tactile-World Action Model that jointly models actions, future visual observations, and tactile dynamics.",
    "abstract": "World action models inherit the predictive capability of world models, enabling action generation to be guided by anticipated future observations. However, they rely primarily on vision and often fail in contact-rich manipulation, where critical cues arise from physical interaction. In this paper, we propose Dream-Tac, a unified Tactile-World Action Model that jointly models actions, future visual observations, and tactile dynamics. Specifically, Dream-Tac introduces (i) contact-gated visuotactile fusion to selectively integrate tactile signals and (ii) a contact-aware attention bias to better regulate cross-modal interactions during manipulation. To support real-time deployment, we further design a dual-level acceleration strategy, reformulating the contact-aware bias to preserve the fused attention path during training and introducing cache-based diffusion acceleration at inference, achieving up to 2.9×\\times faster training and 1.8×\\times faster inference. Across six contact-rich manipulation tasks, Dream-Tac improves action accuracy by 31.7\\% on average, demonstrating the effectiveness of unified visuotactile world modeling.Code is available at https://github.com/LYFCLOUDFAN/Dream-Tac.",
    "submittedDate": "2026-06-07",
    "primaryCategory": "Multimodal / Tactile WAM",
    "secondaryCategories": [
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "DreamTac",
    "arxivUrl": "https://arxiv.org/abs/2606.08737",
    "codeUrls": [
      "https://github.com/LYFCLOUDFAN/Dream-Tac"
    ],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2606.08737",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "联合视频动作建模",
      "多模态触觉音频",
      "高效推理与实时控制"
    ],
    "architecture": "One Model",
    "predictionParadigm": "联合预测",
    "quadrant": "Q1 · One Model × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2606.01205",
    "title": "ImagineUAV: Aerial Vision-Language Navigation via World-Action Modeling and Kinodynamic Planning",
    "authors": "Liu, Xuchen; Huang, Jiawei; Xia, Shihao; Liu, Bingxi; Cui, Jinqiang; Yang, Jiankun",
    "affiliations": "Xuchen Liu, Jinqiang Cui and Jiankun Yang are with Pengcheng Laboratory, Shenzhen, Guangdong, China (e-mail: liuxch,cuijq,jiankun @pcl.ac.cn)",
    "contribution": "To address this, we propose ImagineUAV, an imagination-driven framework leveraging cascaded world-action modeling. With only 1.3B parameters, ImagineUAV outperforms prior VLN and VLA baselines on benchmarks and real-world flights, validating the practicality of imagination-driven aerial navigation.",
    "abstract": "Vision-language navigation (VLN) for UAVs demands grounding free-form instructions into 6-DoF flight under partial observability. While Vision-Language-Action (VLA) models excel at semantic reasoning, they suffer from brittleness due to geometric inconsistency and dynamics mismatch. To address this, we propose ImagineUAV, an imagination-driven framework leveraging cascaded world-action modeling. Instead of direct regression, ImagineUAV employs a latent video diffusion model to generate instruction-conditioned future observations, explicitly imagining environmental evolution, from which 6-DoF motions are inferred via an action extractor. A kinodynamic planner then refines these estimates into collision-free trajectories. Additionally, a step-distilled inference pipeline ensures real-time execution. With only 1.3B parameters, ImagineUAV outperforms prior VLN and VLA baselines on benchmarks and real-world flights, validating the practicality of imagination-driven aerial navigation.",
    "submittedDate": "2026-06-07",
    "primaryCategory": "Navigation / Driving / Domain WAM",
    "secondaryCategories": [
      "3D/4D WAM",
      "Latent / Representation WAM",
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "ImagineUAV",
    "arxivUrl": "https://arxiv.org/abs/2606.01205",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2606.01205",
    "pdfUrl": "https://arxiv.org/pdf/2606.01205",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{ImagineUAV,\n  title={ImagineUAV: Aerial Vision-Language Navigation via World-Action Modeling and Kinodynamic Planning},\n  author={Liu, Xuchen and Huang, Jiawei and Xia, Shihao and Liu, Bingxi and Cui, Jinqiang and Yang, Jiankun},\n  journal={arXiv:2606.01205},\n  year={2026}\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "导航",
      "视觉规划与IDM"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2606.08278",
    "title": "SIMPLE: Simulation-Based Policy Learning and Evaluation for Humanoid Loco-manipulation",
    "authors": "Wei, Songlin; Ni, Zhenhao; Liu, Jie; Zhao, Zhenyu; Ye, Junjie; Jing, Hongyi; Xia, Junkai; Liu, Xiawei; Leong, Michael; Heng, Liang; Huang, Di; Wang, Yue",
    "affiliations": "USC Physical Superintelligence (PSI) Lab; urlcolor=simColor",
    "contribution": "To this end, we present SIMPLE, a unified simulation testbed for humanoid policy learning and evaluation.",
    "abstract": "Humanoid foundation models are advancing faster than we can evaluate them. While real-world testing is expensive and difficult to reproduce, existing simulation benchmarks focus primarily on table-top or wheeled robots. A scalable and reproducible benchmark for whole-body humanoid loco-manipulation remains an open problem. To this end, we present SIMPLE, a unified simulation testbed for humanoid policy learning and evaluation. SIMPLE couples the accurate contact-rich dynamics of MuJoCo with the photorealistic rendering of IsaacSim. It provides a large-scale environment comprising 60 diverse whole-body tasks, 50 indoor scenes, and over 1,000 object assets. To facilitate scalable data collection, the framework integrates two data generation pipelines: automated trajectory generation via motion planning and a low-latency VR teleoperation interface. We further integrate and benchmark mainstream humanoid policies at scale in SIMPLE, including lightweight imitation networks, large vision-language-action (VLA) models, and recent world action models (WAMs). Our experiments reveal a strong correlation between policy performance in simulation and the real world. Furthermore, we demonstrate that policies trained on data collected in SIMPLE can be transferred zero-shot to physical humanoid robots under similar settings, providing a robust and reproducible foundation for humanoid robotics research.",
    "submittedDate": "2026-06-06",
    "primaryCategory": "Navigation / Driving / Domain WAM",
    "secondaryCategories": [
      "Evaluation / Survey / Theory",
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "SIMPLE",
    "arxivUrl": "https://arxiv.org/abs/2606.08278",
    "codeUrls": [
      "https://github.com/physical-superintelligence-lab/SIMPLE"
    ],
    "projectUrl": "https://psi-lab.ai/SIMPLE/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2606.08278",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "人形机器人基准",
      "物理仿真",
      "仿真到真实评测"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2606.08242",
    "title": "Light-WAM: Efficient World Action Models with State-Fusion Action Decoding",
    "authors": "Li, Ziang; Cheng, Dongzhou; Wang, Yibin; Wang, Shiyue; Xu, Xiaoyang; Weng, Lingxuan; Wang, Juan; Wang, Jiaqi",
    "affiliations": "Wuhan University; Shanghai Innovation Institute; Southeast University; Fudan University; East China Normal University",
    "contribution": "We propose Light-WAM, a lightweight World Action Model for efficient robot manipulation. Experiments demonstrate that Light-WAM maintains strong performance on LIBERO and achieves usable multi-task performance on RoboTwin 2.0, while using only 0.44B trainable parameters.",
    "abstract": "World Action Models (WAMs) extend robot policy learning by incorporating future prediction as an additional training objective, encouraging the policy to encode task-relevant temporal structure in its representations. Current WAMs often rely on large-scale generative architectures that incur high training costs and inference latency, making them difficult to deploy as efficient closed-loop policies. We propose Light-WAM, a lightweight World Action Model for efficient robot manipulation. Specifically, it is built with a compact video backbone and performs future-video supervision in a downsampled latent space, reducing the cost of video co-training while retaining its benefits for representation learning. For action prediction, Light-WAM introduces the StateFusionActionExpert, which reads adapted states from multiple backbone layers, fuses them through learned-query pooling, and directly predicts action chunks in a single forward pass. This design provides an efficient interface between video backbone representations and robot actions, avoiding the need for heavy generative action experts. Experiments demonstrate that Light-WAM maintains strong performance on LIBERO and achieves usable multi-task performance on RoboTwin 2.0, while using only 0.44B trainable parameters. It also achieves 72.03ms inference latency with 4.1GiB peak GPU memory and improved training throughput.",
    "submittedDate": "2026-06-06",
    "primaryCategory": "Real-Time / Efficient WAM",
    "secondaryCategories": [
      "Latent / Representation WAM",
      "Memory WAM"
    ],
    "bibtexKey": "LightWAM",
    "arxivUrl": "https://arxiv.org/abs/2606.08242",
    "codeUrls": [
      "https://github.com/L1ziang/Light-WAM"
    ],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2606.08242",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "高效推理与实时控制"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2606.07089",
    "title": "Dreaming when Necessary: Advancing World Action Models with Adaptive Multi-Modal Reasoning",
    "authors": "Tang, Yinzhou; Xu, Jingbo; Shang, Yu; Song, Zihao; Gao, Chen; Wu, Wei; Li, Yong",
    "affiliations": "Tsinghua University",
    "contribution": "Motivated by this observation, we propose \\textbf{AdaWAM}, a world action model with adaptive multimodal reasoning abilities. Experiments on both simulated and real-world embodied tasks show that AdaWAM substantially improves inference efficiency while outperforming state-of-the-art embodied policies.",
    "abstract": "World Action Models (WAMs) offer a promising approach to embodied intelligence, yet existing methods rely heavily on video prediction as action priors and lack adaptive multimodal reasoning, limiting their effectiveness on long-horizon, complex tasks. We observe that WAMs require different multimodal reasoning modes under different execution contexts: textual reasoning is essential during task transitions to guide high-level action prediction, while visual reasoning is critical during fine-grained manipulation for precise control. Motivated by this observation, we propose \\textbf{AdaWAM}, a world action model with adaptive multimodal reasoning abilities. AdaWAM integrates a lightweight dynamic router that autonomously triggers textual or visual reasoning as needed during task execution. Experiments on both simulated and real-world embodied tasks show that AdaWAM substantially improves inference efficiency while outperforming state-of-the-art embodied policies. Codes and demos are available at: https://adawam.github.io/.",
    "submittedDate": "2026-06-05",
    "primaryCategory": "Multimodal / Tactile WAM",
    "secondaryCategories": [
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "DWNAAMMR",
    "arxivUrl": "https://arxiv.org/abs/2606.07089",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2606.07089",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "记忆与长时序",
      "高效推理与实时控制",
      "联合视频动作建模"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "联合预测",
    "quadrant": "Q3 · Dual-system × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2606.05979",
    "title": "World-Language-Action Model for Unified World Modeling, Language Reasoning, and Action Synthesis",
    "authors": "Yang, Yi; Liu, Zhihong; Kou, Siqi; Chen, Yiyang; Hu, Yanzhe; Zhou, Jianbo; Zhao, Boyuan; Wei, Zhijie; Xia, Xiao; Li, Xueqi; Liu, Pengfei; Deng, Zhijie",
    "affiliations": "meta-queries to make the world predictionimplicitlyimpact the action generation so that the former can be disabled; WLA-0 prototype, with 2B active parameters, achieves 40 ms per inference on an NVIDIA RTX 5090. Evaluations across; Code:",
    "contribution": "We propose world-language-action (WLA) models as a new class of embodied foundation models. Our WLA-0 prototype, with 2B active parameters, achieves 40 ms per inference on an NVIDIA RTX 5090.",
    "abstract": "We propose world-language-action (WLA) models as a new class of embodied foundation models. WLA takes textual instructions, images, and robot states as inputs to jointly predict textual subtasks, subgoal images, and robot actions, conjoining the \\emph{world modeling interface} to learn from extensive egocentric videos as in the world-action model (WAM) and the \\emph{language reasoning} capacities to solve complex long-horizon tasks as in vision-language-action (VLA) models. At the core of WLA lies an \\emph{autoregressive (AR)} Transformer backbone, instead of a bidirectional diffusion Transformer as in WAMs, to predict the \\emph{next state}, comprising the \\emph{semantic-level} textual intention and complementary \\emph{fine-grained} physical dynamics. The physical dynamics are supervised by the world modeling objective based on a dedicated World Expert, and are leveraged to ease the characterization of the state-action correlation for the Action Expert. WLA leverages meta-queries to make the world prediction \\emph{implicitly} impact the action generation so that the former can be disabled during inference. The world prediction can also be activated to enable test-time scaling for improved robot control. Our WLA-0 prototype, with 2B active parameters, achieves 40 ms per inference on an NVIDIA RTX 5090. Evaluations across simulated and real-world environments demonstrate that WLA-0 achieves state-of-the-art multi-task and long-horizon learning abilities, e.g., 92.94\\% success rate on RoboTwin2.0 Clean and 56.5\\% success rate on RMBench. WLA-0 also holds the promise to learn novel tasks directly from \\emph{cross-embodiment robot videos} without action annotations.",
    "submittedDate": "2026-06-04",
    "primaryCategory": "General WAM",
    "secondaryCategories": [],
    "bibtexKey": "WLA",
    "arxivUrl": "https://arxiv.org/abs/2606.05979",
    "codeUrls": [
      "https://github.com/SJTU-DENG-Lab/WLA"
    ],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2606.05979",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "潜空间预测与JEPA",
      "记忆与长时序",
      "高效推理与实时控制"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2606.05254",
    "title": "Flash-WAM: Modality-Aware Distillation for World Action Models",
    "authors": "Akbari, Arman; Zhang, Ci; Akbari, Arash; Zhao, Lin; Chen, Yixiao; Chen, Weiwei; Zhang, Xuan; Yuan, Geng; Wang, Yanzhi",
    "affiliations": "1 Northeastern University; 2 University of Georgia; 3 EmbodyX Inc",
    "contribution": "We introduce \\textbf{Flash-WAM}, a modality-aware step-distillation framework inspired by consistency distillation that selects the consistency function for each modality to match its noise regime: a linear-gradient-scaling parametrization for the action stream's low-noise regime, paired with a variance-preserving parametrization for the video stream's high-noise regime, grounded in a structural analysis of the consistency-function family that characterizes the achievable gradient scaling under the consistency boundary condition.",
    "abstract": "World-action models (WAMs) jointly generate future video and robot actions through iterative diffusion, achieving strong performance on manipulation benchmarks but requiring tens of denoising steps, a cost that precludes real-time control. Step distillation has emerged as the natural remedy, but off-the-shelf methods break down in the joint video-action setting because video and action streams use different SNR-shifted noise schedules and reach training with substantially different marginal noise distributions, an asymmetry that single-modality distillation methods cannot accommodate. We introduce \\textbf{Flash-WAM}, a modality-aware step-distillation framework inspired by consistency distillation that selects the consistency function for each modality to match its noise regime: a linear-gradient-scaling parametrization for the action stream's low-noise regime, paired with a variance-preserving parametrization for the video stream's high-noise regime, grounded in a structural analysis of the consistency-function family that characterizes the achievable gradient scaling under the consistency boundary condition. Instantiated on LingBot-VA, Flash-WAM compresses inference to a single step in each modality. On RoboTwin 2.0, this reduces per-chunk latency from 8.18.1 seconds to 348348 ms on NVIDIA L40S, a 23×23{\\times} speedup that enables real-time inference. Flash-WAM preserves task success on simulation benchmarks (85.5%85.5\\% RoboTwin 2.0, 95.7%95.7\\% LIBERO) and substantially recovers real-world performance (60%60\\% average on a Unitree G1 humanoid robot), while naive consistency distillation drops to 24%24\\% at the same step budget.",
    "submittedDate": "2026-06-03",
    "primaryCategory": "Real-Time / Efficient WAM",
    "secondaryCategories": [
      "Evaluation / Survey / Theory",
      "Navigation / Driving / Domain WAM"
    ],
    "bibtexKey": "FlashWAM",
    "arxivUrl": "https://arxiv.org/abs/2606.05254",
    "codeUrls": [
      "https://github.com/NU-World-Model-Embodied-AI/Flash-WAM"
    ],
    "projectUrl": "https://flashwam.github.io/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2606.05254",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "视觉规划与IDM",
      "高效推理与实时控制"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2606.03868",
    "title": "Unified Video-Action Joint Denoising for Dexterous Action and Data Generation",
    "authors": "Wang, Dingrui; Wang, YuAn; Liu, Jinkun; Zhang, Yue; Piccinini, Mattia; Sun, Yu; Betz, Johannes",
    "affiliations": "1 Technical University of Munich; 2 ByteDance; 3 Tsinghua University",
    "contribution": "We propose Donk, a unified video-action denoising model for dexterous hands.",
    "abstract": "Recent world action models leverage video foundation models by aligning broad visual-dynamics priors with executable robot actions. We revisit this alignment from a distributional perspective. Existing formulations typically narrow the aligned prior into an observation-conditioned policy distribution over future actions. In contrast, we keep the distribution broader by modeling the joint space of interaction videos and executable hand trajectories under multiple conditioning regimes. We propose Donk, a unified video-action denoising model for dexterous hands. With language, an initial image, and the initial hand state, Donk samples future videos and bimanual MANO trajectories as an action policy. Without the image condition, the same denoising architecture samples paired video-action rollouts from a text-conditioned distribution, turning the aligned video prior into a data engine. Across action, video, and text-only generation evaluations, Donk improves dexterous trajectory accuracy, preserves strong video fidelity, and produces smooth text-conditioned action rollouts under the same unified training recipe.",
    "submittedDate": "2026-06-02",
    "primaryCategory": "General WAM",
    "secondaryCategories": [],
    "bibtexKey": "UVJDDDG",
    "arxivUrl": "https://arxiv.org/abs/2606.03868",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2606.03868",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "联合视频动作建模",
      "灵巧操作"
    ],
    "architecture": "One Model",
    "predictionParadigm": "联合预测",
    "quadrant": "Q1 · One Model × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2606.03188",
    "title": "GeoSem-WAM: Geometry- and Semantic-Aware World Action Models",
    "authors": "Ma, Fulong; Peng, Daojie; Yue, Wenjun; Cao, Jiahang; Wang, Bintao; Zhang, Qiang; Ma, Jun",
    "affiliations": "Not identified",
    "contribution": "To address this, we propose a structured world modeling framework that enhances latent representations through geometric and semantic supervision. Alongside future RGB prediction, our model introduces two auxiliary prediction branches for future geometry and semantic representations, enabling it to jointly capture scene dynamics, spatial geometry, and semantic context within a unified latent space.",
    "abstract": "Recent World Action Models (WAMs) have demonstrated impressive capabilities in embodied decision-making. However, whether their effectiveness stems from explicit future imagination during inference or representation learning induced by predictive training remains an open question. Emerging evidence suggests the primary advantage lies in learning robust latent representations rather than generating future observations at test time. Nevertheless, existing WAMs mainly rely on RGB-based future prediction, which provides limited structural and spatial understanding of complex environments. To address this, we propose a structured world modeling framework that enhances latent representations through geometric and semantic supervision. Alongside future RGB prediction, our model introduces two auxiliary prediction branches for future geometry and semantic representations, enabling it to jointly capture scene dynamics, spatial geometry, and semantic context within a unified latent space. Crucially, our approach preserves efficient inference by avoiding explicit future rollout or video generation at test time. Extensive experiments show that incorporating structured world supervision consistently improves action prediction accuracy, scene understanding, and robustness under challenging embodied scenarios, highlighting its potential for advancing scalable and efficient WAMs.",
    "submittedDate": "2026-06-02",
    "primaryCategory": "3D/4D WAM",
    "secondaryCategories": [
      "Latent / Representation WAM",
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "GeoSemWAM",
    "arxivUrl": "https://arxiv.org/abs/2606.03188",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2606.03188",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "三维多视角建模",
      "高效推理与实时控制"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2606.01955",
    "title": "WALL-WM: Carving World Action Modeling at the Event Joints",
    "authors": "Shalfun Li; Victor Yao; Charles Yang; Truth Qu; Regis Cheng; Ryan Yu; Howard Lu; Newton Von; Vincent Chen; Yohann Tang; Maeve Zhang; Ellie Ma; Gody Li; Sage Yang; Lorien Shu; J. W. Gao; Ethan Chen; Colin Ye; Yu Sun; Elise Mon; PS Zhang; Neo Li; Lily Li; James Wang; Ping Yang; Chris Pan; Lucy Liang; Hang Su; Roy Gan; Hao Wang; Qian Wang",
    "affiliations": "Not identified",
    "contribution": "Experiments show that WALL-WM generalizes broadly across language, scenes, and tasks, achieving state-of-the-art performance in large-scale real-world generalization evaluation.",
    "abstract": "WALL-WM is a World Action Model that shifts video-action learning from chunk-centric optimization to event-grounded Vision-Language-Action pretraining, using semantically coherent action events as the atomic unit of learning. Existing WAMs commonly initialize from multimodal or video foundation models and then optimize fixed-length action chunks conditioned directly on the current observation and instruction. Although convenient, this chunk-centric formulation creates a fundamental granularity mismatch. Language describes semantic goals and events, vision evolves through continuous scene dynamics, and actions operate at control-level timescales; forcing all three into the same fixed-length prediction window turns VLA training into short-horizon correlation fitting. WALL-WM addresses this mismatch by organizing both supervision and data around semantic events. Specifically, it pairs event-grounded VLA pretraining with a data ecosystem built from event-level captions and cluster-balanced sampling, enabling scalable learning over diverse behaviors, scenes, and task structures. From the same event-pretrained backbone, WALL-WM supports two complementary inference modes. The event mode consumes next-event descriptions and enables variable-length execution chunks, while the unified mode uses a VLM with Staircase Decoding to condition conventional fixed-length chunk inference while preserving a gradient-continuous VLA path. Together with Muon-optimizer-based large-scale pretraining infrastructure, WALL-WM provides a practical scale-up recipe for general-purpose WAMs. Experiments show that WALL-WM generalizes broadly across language, scenes, and tasks, achieving state-of-the-art performance in large-scale real-world generalization evaluation.",
    "submittedDate": "2026-06-01",
    "primaryCategory": "Multimodal / Tactile WAM",
    "secondaryCategories": [
      "Evaluation / Survey / Theory"
    ],
    "bibtexKey": "WALLWM",
    "arxivUrl": "https://arxiv.org/abs/2606.01955",
    "codeUrls": [
      "https://github.com/X-Square-Robot/wall-wm"
    ],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2606.01955",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "视觉规划与IDM",
      "记忆与长时序",
      "泛化与动作对齐"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2605.13548",
    "title": "AttenA+: Rectifying Action Inequality in Robotic Foundation Models",
    "authors": "Peng, Daojie; Ma, Fulong; Cao, Jiahang; Zhang, Qiang; Xie, Xupeng; Guo, Jian; Luo, Ping; Luo, Andrew F.; Zhou, Boyu; Ma, Jun",
    "affiliations": "IDEA Research SUSTech X-Humaniod",
    "contribution": "To rectify this, we introduce AttenA+, an architecture-agnostic framework that prioritizes kinematically critical segments via velocity-driven action attention. Extensive experiments demonstrate that AttenA+ significantly elevates the ceilings of current state-of-the-art models.",
    "abstract": "Existing robotic foundation models, while powerful, are predicated on an implicit assumption of temporal homogeneity: treating all actions as equally informative during optimization. This \"flat\" training paradigm, inherited from language modeling, remains indifferent to the underlying physical hierarchy of manipulation. In reality, robot trajectories are fundamentally heterogeneous, where low-velocity segments often dictate task success through precision-demanding interactions, while high-velocity motions serve as error-tolerant transitions. Such a misalignment between uniform loss weighting and physical criticality fundamentally limits the performance of current Vision-Language-Action (VLA) models and World-Action Models (WAM) in complex, long-horizon tasks. To rectify this, we introduce AttenA+, an architecture-agnostic framework that prioritizes kinematically critical segments via velocity-driven action attention. By reweighting the training objective based on the inverse velocity field, AttenA+ naturally aligns the model's learning capacity with the physical demands of manipulation. As a plug-and-play enhancement, AttenA+ can be integrated into existing backbones without structural modifications or additional parameters. Extensive experiments demonstrate that AttenA+ significantly elevates the ceilings of current state-of-the-art models. Specifically, it improves OpenVLA-OFT to 98.6% (+1.5%) on the Libero benchmark and pushes FastWAM to 92.4% (+0.6%) on RoboTwin 2.0. Real-world validation on a Franka manipulator further showcases its robustness and cross-task generalization. Our work suggests that mining the intrinsic structural priors of action sequences offers a highly efficient, physics-aware complement to standard scaling laws, paving a new path for general-purpose robotic control.",
    "submittedDate": "2026-06-01",
    "primaryCategory": "Real-Time / Efficient WAM",
    "secondaryCategories": [
      "Evaluation / Survey / Theory"
    ],
    "bibtexKey": "AttenA",
    "arxivUrl": "https://arxiv.org/abs/2605.13548",
    "codeUrls": [
      "https://github.com/DaojiePENG/AttenA-Plus"
    ],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2605.13548",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "VLA",
    "subcategories": [
      "训练优化与蒸馏",
      "动作策略基础"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2606.01095",
    "title": "Beyond Task Success: Behavioral and Representational Diagnostics for WAM and VLA",
    "authors": "Mai, Hung; Zhu, Bin; Do, Tuan",
    "affiliations": "National Economics University, Vietnam; N2TP Technology; Singapore Management University; Phenikaa University, Vietnam",
    "contribution": "We introduce a model-agnostic diagnostic framework that compares WAMs and VLAs through two complementary lenses: behavioral rollout analysis and sparse-autoencoder-based feature analysis.",
    "abstract": "Vision-language-action (VLA) policies and World-Action Models (WAM) represent two increasingly important paradigms for robotic manipulation. However, it remains unclear whether future prediction in WAMs leads to behaviorally meaningful improvements beyond final task success. In this paper, we ask whether WAMs merely add future prediction, or whether they change robot behavior and internal representations in ways that are actionable for control. We introduce a model-agnostic diagnostic framework that compares WAMs and VLAs through two complementary lenses: behavioral rollout analysis and sparse-autoencoder-based feature analysis. The behavioral protocol measures action dynamics consistency, target-object progress, distractor disturbance, and runtime cost. The feature-space protocol characterizes internal representations as memorized, reactive, or predictive, revealing whether models encode future-oriented structure. Across LIBERO and RoboTwin2.0, we evaluate 7 policies spanning direct VLAs and joint, sequential, and auxiliary WAMs. Our results show that success alone hides key differences: WAMs often improve object-level behavior and target selectivity, but their gains depend on architecture and incur higher inference cost. Sequential WAMs show the clearest predictive structure, while auxiliary and joint WAMs respectively compress or entangle future information. These findings suggest future directions for WAMs design to preserve behaviorally actionable future representations for efficient manipulation.",
    "submittedDate": "2026-05-31",
    "primaryCategory": "Real-Time / Efficient WAM",
    "secondaryCategories": [
      "Evaluation / Survey / Theory"
    ],
    "bibtexKey": "BTSBRDWV",
    "arxivUrl": "https://arxiv.org/abs/2606.01095",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2606.01095",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "评估指标（Metrics）",
    "subcategories": [
      "行为与表征诊断",
      "评估指标与协议"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2605.28544",
    "title": "DriveWAM: Video Generative Priors Enable Scalable World-Action Modeling for Autonomous Driving",
    "authors": "Shi, Chen; Xu, Jinrui; Shi, Shaoshuai; Sheng, Kehua; Zhang, Bo; Jiang, Li",
    "affiliations": "1 The Chinese University of Hong Kong, Shenzhen 2 Voyager Research, Didi Chuxing",
    "contribution": "We present DriveWAM, a driving world-action model that adapts a pretrained video diffusion transformer into an autoregressive video-action policy. To incorporate high-level scene understanding, we introduce scene-evolving driving guidance, where a frozen VLM produces chunk-specific semantic intent to guide video-action generation.",
    "abstract": "Pretrained foundation models have become an important basis for end-to-end autonomous driving. In contrast to vision-language models pretrained primarily on static image-text pairs, video generative models capture temporal dynamics and motion priors that are naturally suited for driving. We present DriveWAM, a driving world-action model that adapts a pretrained video diffusion transformer into an autoregressive video-action policy. DriveWAM organizes video and action streams into a unified temporal token sequence and trains them under a joint flow-matching objective, preserving the pretrained video-generation architecture while adapting its large-scale video priors to action generation. To incorporate high-level scene understanding, we introduce scene-evolving driving guidance, where a frozen VLM produces chunk-specific semantic intent to guide video-action generation. To keep long-horizon rollout bounded, we further introduce selective KV memory, which maintains bounded modality-aware video and action memory pools through relevance-redundancy cache selection at inference time. Experiments on NAVSIM and the PhysicalAI-Autonomous-Vehicles benchmark show that DriveWAM achieves strong planning performance, and a data-scaling study from 4k to 100k driving clips further confirms the scaling potential of world-action modeling for end-to-end autonomous driving.",
    "submittedDate": "2026-05-27",
    "primaryCategory": "Navigation / Driving / Domain WAM",
    "secondaryCategories": [
      "Evaluation / Survey / Theory",
      "Memory WAM",
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "DriveWAM",
    "arxivUrl": "https://arxiv.org/abs/2605.28544",
    "codeUrls": [
      "https://github.com/chenshi3/DriveWAM"
    ],
    "projectUrl": "https://chenshi3.github.io/drivewam.github.io/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2605.28544",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "自动驾驶",
      "视觉规划与IDM",
      "记忆与长时序"
    ],
    "architecture": "One Model",
    "predictionParadigm": "IDM",
    "quadrant": "Q2 · One Model × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2605.25829",
    "title": "OASIS: Observation-Action Space Alignment via SE(3) Trajectory Prediction for Robotic Manipulation",
    "authors": "Chen, Xinzhe; Ren, Sihua; Huang, Liqi; Sun, Haowen; Li, Mingyang; Chen, Xingyu; Liu, Zeyang; Lan, Xuguang",
    "affiliations": "National Key Laboratory of Human-Machine Hybrid Augmented Intelligence; Institute of Artificial Intelligence and Robotics, Xi'an Jiaotong University",
    "contribution": "We propose OASIS, a visuomotor policy that aligns the intermediate representation with the action space via SE(3)SE(3) end-effector trajectory prediction. Across simulation and real-world experiments, OASIS outperforms VLA and WAM baselines in success rate and out-of-distribution generalization.",
    "abstract": "Recent vision-language-action (VLA) models and world action models (WAMs) advance robotic manipulation by enriching intermediate representations with auxiliary spatial features or future visual-state prediction. However, these representations largely remain within the observation space and do not share the rigid-body geometry of the action space, forcing the action decoder to implicitly recover this geometry. We propose OASIS, a visuomotor policy that aligns the intermediate representation with the action space via SE(3)SE(3) end-effector trajectory prediction. OASIS couples a 3D-aware feature encoder that fuses vision-language and metric-depth features with an SE(3)SE(3) trajectory predictor that produces a camera-frame end-effector trajectory. Conditioned on the predictor's pose-supervised hidden states, the action decoder generates action chunks consistent with rigid-body motion. Across simulation and real-world experiments, OASIS outperforms VLA and WAM baselines in success rate and out-of-distribution generalization. Our project page is available at https://npuhandsome.github.io/OASIS_web.",
    "submittedDate": "2026-05-25",
    "primaryCategory": "3D/4D WAM",
    "secondaryCategories": [],
    "bibtexKey": "OASIS",
    "arxivUrl": "https://arxiv.org/abs/2605.25829",
    "codeUrls": [],
    "projectUrl": "https://npuhandsome.github.io/OASIS_web/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2605.25829",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "VLA",
    "subcategories": [
      "未来表征辅助VLA",
      "扩散与流匹配VLA"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2605.23856",
    "title": "Point Tracking Improves World Action Models",
    "authors": "Guan, Jiarui; Zhao, Wenshuai; Pei, Yue; Chen, Ziliang; Solin, Arno; Kannala, Juho",
    "affiliations": "Aalto University; ELLIS Institute Finland; University of Oulu; Sun Yat-sen University; Peng Cheng Laboratory; Beihang University",
    "contribution": "We propose JOPAT, a JOint Pixel-And-Track World-Action Model that predicts latent visual observations, 2D point tracks with visibility, and actions in a single denoising diffusion transformer.",
    "abstract": "Robot policy learning benefits from world-action models that capture environment dynamics, but pixel-level prediction entangles dynamics with nuisance factors such as lighting and texture, making learned representations vulnerable to task-irrelevant visual variation. We propose JOPAT, a JOint Pixel-And-Track World-Action Model that predicts latent visual observations, 2D point tracks with visibility, and actions in a single denoising diffusion transformer. The key insight is that tracks provide an explicit representation of motion that captures long-horizon dynamics and remains robust under occlusion or partial out-of-frame motion, offering greater utility than modeling pixel appearance alone. On LIBERO and real-world LeRobot tasks, JOPAT improves over pixel-based baselines, with the largest gains on long-horizon tasks involving occlusion, object interaction, and off-screen motion.",
    "submittedDate": "2026-05-22",
    "primaryCategory": "Latent / Representation WAM",
    "secondaryCategories": [],
    "bibtexKey": "JOPAT",
    "arxivUrl": "https://arxiv.org/abs/2605.23856",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2605.23856",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "联合视频动作建模",
      "记忆与长时序",
      "泛化与动作对齐"
    ],
    "architecture": "One Model",
    "predictionParadigm": "联合预测",
    "quadrant": "Q1 · One Model × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2605.18556",
    "title": "Key-Gram: Extensible World Knowledge for Embodied Manipulation",
    "authors": "Fan, Jingjing; Li, Siyuan; Ren, Botao; Deng, Zhidong",
    "affiliations": "Department of Computer Science and Technology; Department of Automation; Tsinghua University",
    "contribution": "In this paper, we introduce Key-Gram, a conditional-memory framework that separates language-derived world knowledge from visual-state reasoning for embodied control.",
    "abstract": "Embodied control increasingly requires models to follow compositional language instructions while reasoning over dynamic visual states. However, current vision-language-action policies and world-action models often couple linguistic knowledge with visual computation in a shared backbone or conditioning pathway, leading to modality competition and making knowledge extension dependent on backbone updates. In this paper, we introduce Key-Gram, a conditional-memory framework that separates language-derived world knowledge from visual-state reasoning for embodied control. At its core is a memory module that decomposes an instruction into task-specific key-grams, retrieves static linguistic priors through deterministic hashed lookup, and injects the retrieved entries into selected hidden layers through context-aware gating and lightweight convolutional fusion. This design allows the backbone to devote its main capacity to visual reasoning and action inference, while reusable instruction knowledge is stored in an extensible external memory. The logical memory table can be conveniently partitioned during training and, due to its O(1)O(1) lookup pattern, efficiently placed on host memory during inference. Across RoboTwin2.0, LIBERO/LIBERO-Plus, and real-world dual-arm manipulation, Key-Gram consistently improves both π0π_{0} and π0.5π_{0.5} backbones, with average relative gains of 29.5%/9.9%29.5\\%/9.9\\% on RoboTwin2.0, 35.8%/4.5%35.8\\%/4.5\\% on LIBERO-Plus transfer without target-domain fine-tuning, and 15.4%/8.1%15.4\\%/8.1\\% on real-world long-horizon tasks. These results demonstrate that externalized linguistic memory provides an effective and extensible mechanism for improving compositional grounding, transfer, and real-world manipulation.",
    "submittedDate": "2026-05-18",
    "primaryCategory": "Memory WAM",
    "secondaryCategories": [
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "KeyGram",
    "arxivUrl": "https://arxiv.org/abs/2605.18556",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2605.18556",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "VLA",
    "subcategories": [
      "VLA后训练与数据增强",
      "分层与双系统VLA"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2605.17077",
    "title": "How to Instruct Your Robot: Dense Language Annotations Power Robot Policy Learning",
    "authors": "Kim, Bosung; Wang, Ruiyi; Acuna, David; Jung, Jaehun; Trevithick, Alexander; Cui, Brandon; Choi, Yejin; Ammanabrolu, Prithviraj",
    "affiliations": "University of California, San Diego; NVIDIA",
    "contribution": "We introduce DeMiAn (Dense Multi-aspect Annotation), a two-stage approach that first re-labels demonstration segments with VLM-generated annotations along four complementary aspects: physical motion, scene composition, arm pose, and reasoning.",
    "abstract": "Scaling robot policy learning is bottlenecked by the cost of collecting demonstrations, while language annotations for existing demonstrations are comparatively cheap. We study language density as a lever for extracting more signal from a fixed robot or egocentric-video corpus. We introduce DeMiAn (Dense Multi-aspect Annotation), a two-stage approach that first re-labels demonstration segments with VLM-generated annotations along four complementary aspects: physical motion, scene composition, arm pose, and reasoning. A learned instructor then maps a task description and initial scene snapshot to a task-appropriate annotation at deployment, running asynchronously so generation latency is hidden behind policy execution. Across over 1M robot manipulation clips and 50K EgoVerse human-egocentric videos, DeMiAn improves both a vision-language-action policy and a video-based world-action model without collecting new demonstrations. On RoboCasa, the instructor raises success by 5 points over a task-only baseline and comes within 3 points of a per-task oracle. No fixed annotation aspect dominates across tasks, showing that selecting the right dense language matters. DeMiAn also improves composite-task and out-of-distribution performance, and shifts the compute-performance frontier in both mid-training and post-training after accounting for annotation-generation FLOPs. These results position dense re-annotation as a practical scaling lever for robot policy learning.",
    "submittedDate": "2026-05-16",
    "primaryCategory": "General WAM",
    "secondaryCategories": [],
    "bibtexKey": "DeMiAn",
    "arxivUrl": "https://arxiv.org/abs/2605.17077",
    "codeUrls": [],
    "projectUrl": "https://www.cs.toronto.edu/~davidj/publication/instruct-robot/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2605.17077",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "数据集",
    "subcategories": [
      "语言标注与再标注",
      "机器人示范与操作数据",
      "人类第一视角数据"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2605.15964",
    "title": "WorldVLN: Autoregressive World Action Model for Aerial Vision-Language Navigation",
    "authors": "Zhao, Baining; Xu, Jiacheng; Feng, Weicheng; Zhang, Xin; Wang, Zhaolu; Wang, Haoyang; Ji, Shilong; Wang, Ziyou; Fang, Jianjie; Zheng, Zhiheng; Zhang, Weichen; Shang, Yu; Wu, Wei; Gao, Chen; Chen, Xinlei; Li, Yong",
    "affiliations": "1 Tsinghua University; 2 Shandong University; 4 Beijing Institute of Technology; 5 Northeastern University",
    "contribution": "To this end, we propose WorldVLN, the first autoregressive world action model for aerial VLN. On public outdoor and indoor benchmarks, WorldVLN consistently outperforms existing Vision-Language-Action baselines with 12\\%+ success-rate gains and larger advantages on challenging cases.",
    "abstract": "Aerial vision-language navigation (VLN) requires agents to follow natural-language instructions through closed-loop perception and action in 3D environments. We argue that aerial VLN can be formulated as a prediction-driven world-action problem: the agent should anticipate latent world evolution and act according to the predicted consequences. To this end, we propose WorldVLN, the first autoregressive world action model for aerial VLN. Unlike full-sequence video-generation world models that generate an entire visual clip, WorldVLN adapts a latent autoregressive video backbone to predict short-horizon world-state transitions and directly decodes them into executable waypoint actions. After each action segment is executed, newly received observations are encoded back into the autoregressive context, enabling closed-loop world-action prediction. We further introduce a two-stage training framework that first grounds the video prior in instruction-conditioned navigation dynamics and then develops Action-aware GRPO, the first reinforcement learning method tailored to autoregressive WAMs, to optimize waypoint decisions through their downstream rollout consequences. On public outdoor and indoor benchmarks, WorldVLN consistently outperforms existing Vision-Language-Action baselines with 12\\%+ success-rate gains and larger advantages on challenging cases. It further transfers zero-shot to real drone deployment, suggesting that the proposed WorldVLN offers a promising route for spatial action tasks. Demos and code are available at https://embodiedcity.github.io/WorldVLN/.",
    "submittedDate": "2026-05-15",
    "primaryCategory": "Navigation / Driving / Domain WAM",
    "secondaryCategories": [
      "3D/4D WAM",
      "Latent / Representation WAM",
      "WAM + RL"
    ],
    "bibtexKey": "WorldVLN",
    "arxivUrl": "https://arxiv.org/abs/2605.15964",
    "codeUrls": [
      "https://github.com/EmbodiedCity/WorldVLN.code"
    ],
    "projectUrl": "https://embodiedcity.github.io/WorldVLN/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2605.15964",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "导航",
      "视觉规划与IDM",
      "策略后训练与WM-RL"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2605.12090",
    "title": "World Action Models: The Next Frontier in Embodied AI",
    "authors": "Wang, Siyin; Shi, Junhao; Fu, Zhaoyang; He, Xinzhe; Liu, Feihong; Yang, Chenchen; Zhou, Yikang; Fei, Zhaoye; Gong, Jingjing; Fu, Jinlan; Shou, Mike Zheng; Huang, Xuanjing; Qiu, Xipeng; Jiang, Yu-Gang",
    "affiliations": "1 Fudan University; 2 Shanghai Innovation Institute; 3 National University of Singapore",
    "contribution": "Vision-Language-Action (VLA) models have achieved strong semantic generalization for embodied policy learning, yet they learn reactive observation-to-action mappings without explicitly modeling how the physical world evolves under intervention. A growing body of work addresses this limitation by integrating world models, predictive models of environment dynamics, into the action generation pipeline.",
    "abstract": "Vision-Language-Action (VLA) models have achieved strong semantic generalization for embodied policy learning, yet they learn reactive observation-to-action mappings without explicitly modeling how the physical world evolves under intervention. A growing body of work addresses this limitation by integrating world models, predictive models of environment dynamics, into the action generation pipeline. We term this emerging paradigm World Action Models (WAMs): embodied foundation models that unify predictive state modeling with action generation, targeting a joint distribution over future states and actions rather than actions alone. However, the literature remains fragmented across architectures, learning objectives, and application scenarios, lacking a unified conceptual framework. We formally define WAMs and disambiguate them from related concepts, and trace the foundations and early integration of VLA and world model research that gave rise to this paradigm. We organize existing methods into a structured taxonomy of Cascaded and Joint WAMs, with further subdivision by generation modality, conditioning mechanism, and action decoding strategy. We systematically analyze the data ecosystem fueling WAMs development, spanning robot teleoperation, portable human demonstrations, simulation, and internet-scale egocentric video, and synthesize emerging evaluation protocols organized around visual fidelity, physical commonsense, and action plausibility. Overall, this survey provides the first systematic account of the WAMs landscape, clarifies key architectural paradigms and their trade-offs, and identifies open challenges and future opportunities for this rapidly evolving field.",
    "submittedDate": "2026-05-12",
    "primaryCategory": "Evaluation / Survey / Theory",
    "secondaryCategories": [],
    "bibtexKey": "NFEA",
    "arxivUrl": "https://arxiv.org/abs/2605.12090",
    "codeUrls": [
      "https://github.com/OpenMOSS/Awesome-WAM"
    ],
    "projectUrl": "https://openmoss.ai/Awesome-WAM/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2605.12090",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "Related resources",
    "subcategories": [
      "综述与技术资源"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2605.11550",
    "title": "The DAWN of World-Action Interactive Models",
    "authors": "Lu, Hongbo; Yao, Liang; He, Chenghao; Wang, Haoyu; Gu, Xiang; Li, Xianfei; Liao, Wenlong; He, Tao; Peng, Pai",
    "affiliations": "1COWARobot Co. Ltd 2Shanghai Jiao Tong University 3Hohai University",
    "contribution": "Experiments show that DAWN achieves strong planning performance and favorable safety-related results across multiple autonomous driving benchmarks.",
    "abstract": "A plausible scene evolution depends on the maneuver being considered, while a good maneuver depends on how the scene may evolve. Existing World Action Models (WAMs) largely miss this reciprocity, treating world prediction and action generation as either isolated parallel branches or rigid predict-then-plan pipelines. We formalize this perspective as World-Action Interactive Models (WAIMs), and instantiate it in autonomous driving with \\textbf{DAWN} (\\textbf{D}enoising \\textbf{A}ctions and \\textbf{W}orld i\\textbf{N}teractive model), a simple yet strong latent generative baseline. DAWN operates in a compact semantic latent space and couples a \\emph{World Predictor} with a \\emph{World-Conditioned Action Denoiser}: the predicted world hypothesis conditions action denoising, while the denoised action hypothesis is fed back to update the world prediction, so that both are recursively refined during inference. Rather than eliminating test-time world evolution altogether or rolling out the full future in pixel space, DAWN performs a short explicit latent rollout that is sufficient to support long-horizon trajectory generation in complex interactive scenes. Experiments show that DAWN achieves strong planning performance and favorable safety-related results across multiple autonomous driving benchmarks. More broadly, our results suggest that interactive world-action generation is a principled path toward truly actionable world models.",
    "submittedDate": "2026-05-12",
    "primaryCategory": "Latent / Representation WAM",
    "secondaryCategories": [
      "Navigation / Driving / Domain WAM",
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "WAIMs",
    "arxivUrl": "https://arxiv.org/abs/2605.11550",
    "codeUrls": [
      "https://github.com/COOWAI/DAWN"
    ],
    "projectUrl": "https://cowarobot-ai.github.io/DAWN/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2605.11550",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "自动驾驶",
      "潜空间预测与JEPA",
      "联合视频动作建模"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "联合预测",
    "quadrant": "Q3 · Dual-system × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2605.10942",
    "title": "HarmoWAM: Harmonizing Generalizable and Precise Manipulation via Adaptive World Action Models",
    "authors": "Feng, Qiuxuan; Yu, Jiale; Liu, Jiaming; Jia, Yueru; Wu, Zhuangzhe; Chen, Hao; Qian, Zezhong; Gu, Shuo; Jia, Peng; Ma, Siwei; Zhang, Shanghang",
    "affiliations": "State Key Laboratory of Multimedia Information Processing, School of Computer Science; Peking University; Simplexity Robotics; The Chinese University of Hong Kong 0.2cm",
    "contribution": "Motivated by these findings, we propose HarmoWAM, an end-to-end WAM that fully leverages a world model to unify predictive and reactive control, enabling both generalizable transit and precise manipulation. Notably, HarmoWAM achieves strong zero-shot generalization across these scenarios, significantly outperforming prior state-of-the-art VLA models and WAMs by margins of 33% and 29%, respectively.",
    "abstract": "World Action Models (WAMs) have emerged as a promising paradigm for robot control by modeling physical dynamics. Current WAMs generally follow two paradigms: the \"Imagine-then-Execute\" approach, which uses video prediction to infer actions via inverse dynamics, and the \"Joint Modeling\" approach, which jointly models actions and video representations. Based on systematic experiments, we observe a fundamental trade-off between these paradigms: the former explicitly leverages world models for generalizable transit but lacks interaction precision, whereas the latter enables fine-grained, temporally coherent action generation but is constrained by the exploration space of the training distribution. Motivated by these findings, we propose HarmoWAM, an end-to-end WAM that fully leverages a world model to unify predictive and reactive control, enabling both generalizable transit and precise manipulation. Specifically, the world model provides spatio-temporal physical priors that condition two complementary action experts: a predictive expert that leverages latent dynamics for iterative action generation, and a reactive expert that directly infers actions from predicted visual evolution. To enable adaptive coordination, a Process-Adaptive Gating Mechanism is proposed to automatically determine the timing and location of switching between them. This allows the world model to drive the reactive expert to expand the exploration space and the predictive expert to perform precise interactions across different stages of a task. For evaluation, we construct three training-unseen test environments across six real-world robotic tasks, covering variations in background, position, and object semantics. Notably, HarmoWAM achieves strong zero-shot generalization across these scenarios, significantly outperforming prior state-of-the-art VLA models and WAMs by margins of 33% and 29%, respectively.",
    "submittedDate": "2026-05-11",
    "primaryCategory": "Latent / Representation WAM",
    "secondaryCategories": [
      "Evaluation / Survey / Theory"
    ],
    "bibtexKey": "HarmoWAM",
    "arxivUrl": "https://arxiv.org/abs/2605.10942",
    "codeUrls": [],
    "projectUrl": "https://elbb-yu.github.io/HarmoWAM/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2605.10942",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "视觉规划与IDM",
      "泛化与动作对齐"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2605.06222",
    "title": "When to Trust Imagination: Adaptive Action Execution for World Action Models",
    "authors": "Wang, Rui; Zhang, Yue; Lin, Jiehong; Luo, Kuncheng; Wang, Jianan; Wang, Zhongrui; Qi, Xiaojuan",
    "affiliations": "1 Southern University of Science and Technology, Shenzhen, China; 2 The University of Hong Kong, Hong Kong, China",
    "contribution": "In this work, we formulate adaptive WAM execution as a future-reality verification problem: the robot should execute longer when the WAM-predicted future remains reliable, and replan earlier when reality deviates from imagination. To this end, we propose Future Forward Dynamics Causal Attention (FFDC), a lightweight verifier that jointly reasons over predicted future actions, predicted visual dynamics, real observations, and language instructions to estimate whether the remaining action rollout can still be trusted.",
    "abstract": "World Action Models (WAMs) have recently emerged as a promising paradigm for robotic manipulation by jointly predicting future visual observations and future actions. However, current WAMs typically execute a fixed number of predicted actions after each model inference, leaving the robot blind to whether the imagined future remains consistent with the actual physical rollout. In this work, we formulate adaptive WAM execution as a future-reality verification problem: the robot should execute longer when the WAM-predicted future remains reliable, and replan earlier when reality deviates from imagination. To this end, we propose Future Forward Dynamics Causal Attention (FFDC), a lightweight verifier that jointly reasons over predicted future actions, predicted visual dynamics, real observations, and language instructions to estimate whether the remaining action rollout can still be trusted. FFDC enables adaptive action chunk sizes as an emergent consequence of prediction-observation consistency, preserving the efficiency of long-horizon execution while restoring responsiveness in contact-rich or difficult phases. We further introduce Mixture-of-Horizon Training to improve long-horizon trajectory coverage for adaptive execution. Experiments on the RoboTwin benchmark and in the real world demonstrate that our method achieves a strong robustness-efficiency trade-off: on RoboTwin, it reduces WAM forward passes by 69.10% and execution time by 34.02%, while improving success rate by 2.54% over the short-chunk baseline; in real-world experiments, it improves success rate by 35%.",
    "submittedDate": "2026-05-09",
    "primaryCategory": "Real-Time / Efficient WAM",
    "secondaryCategories": [
      "Evaluation / Survey / Theory"
    ],
    "bibtexKey": "FFDC",
    "arxivUrl": "https://arxiv.org/abs/2605.06222",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2605.06222",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "记忆与长时序",
      "高效推理与实时控制",
      "策略后训练与WM-RL"
    ],
    "architecture": "不适用",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2605.08638",
    "title": "Geometry Guided Self-Consistency for Physical AI",
    "authors": "Dai, Yinwei; Chen, Zhuofu; Yang, Lijie; Netravali, Ravi",
    "affiliations": "Princeton University",
    "contribution": "State-of-the-art physical AI models generate a chunk of actions per inference through diffusion or flow matching, iteratively refining an initial noise sample into an action trajectory. We introduce KeyStone, an inference-time self-consistency method for diffusion-based action generation that draws KK candidate action chunks in parallel from a shared model context, clusters them in continuous action space, and returns the medoid of the largest cluster -- no additional model required.",
    "abstract": "State-of-the-art physical AI models generate a chunk of actions per inference through diffusion or flow matching, iteratively refining an initial noise sample into an action trajectory. Because this inference process is inherently stochastic, committing to a single trajectory per round is brittle, and this brittleness compounds across the many sequential rounds that comprise a complete episode. We introduce KeyStone, an inference-time self-consistency method for diffusion-based action generation that draws KK candidate action chunks in parallel from a shared model context, clusters them in continuous action space, and returns the medoid of the largest cluster -- no additional model required. Two properties make this practical. First, the compact nature of action trajectories makes diffusion inference memory-bandwidth bound, leaving spare compute capacity to run KK chains in parallel with no additional wall-clock latency. Second, unlike token or pixel spaces where distance carries no semantic meaning and selection requires a learned judge, action chunks are geometrically structured such that Euclidean distance directly reflects physical similarity, making selection principled and judge-free. Across diverse vision-language-action models (VLAs) and world-action models (WAMs), KeyStone improves task success rates by up to \\textbf{13.3\\%} over single-trajectory sampling with negligible latency overhead, while having on par accuracy with model-based selectors at no training cost. We open source KeyStone at https://github.com/dywsjtu/keystone.",
    "submittedDate": "2026-05-08",
    "primaryCategory": "3D/4D WAM",
    "secondaryCategories": [
      "Memory WAM",
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "KeyStone",
    "arxivUrl": "https://arxiv.org/abs/2605.08638",
    "codeUrls": [
      "https://github.com/dywsjtu/keystone"
    ],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2605.08638",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "VLA",
    "subcategories": [
      "动作策略基础",
      "理论与规划"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2605.07794",
    "title": "NoiseGate: Learning Per-Latent Timestep Schedules as Information Gating in World Action Models",
    "authors": "Huang, Wen; Sun, Haoran; Guo, Yongjian; Ma, Yunxuan; Li, Haoran; Long, Jing; Mo, Zhouying; Guan, Zhong; Guo, Yucheng; Di, Shuai; Xiong, Junwu",
    "affiliations": "1 Tsinghua University, 2 Peking University, 3 JDT AI Infra, 4 Tianjin University",
    "contribution": "We propose \\textbf{NoiseGate}, which combines independent per-latent timestep sampling during backbone training, a lightweight Gating Policy Network that emits per-latent time increments during denoising, and task-reward optimization that trains the schedule policy without hand-crafted shape priors.",
    "abstract": "World Action Models (WAMs) are an emerging family of policies that tie robot action generation to future-observation modeling. In this work, we focus on the joint video--action modeling paradigm, where actions and imagined future observations are co-generated along a shared denoising or flow trajectory, so that perception, prediction, and control are coupled within one generative process. Existing WAMs typically realize this paradigm with a Mixture-of-Transformers (MoT), where video and action tokens interact through shared self-attention. This architecture can in principle assign a separate timestep tft_f to each predicted latent frame, yet current systems collapse this degree of freedom onto a single shared scalar tt. Under the noise-as-masking view of Diffusion Forcing, this shared schedule imposes the unjustified prior that every predicted latent is equally reliable for action generation. We instead view the per-latent schedule as a \\emph{learnable information-gating policy}: by changing a latent frame's noise level, the policy modulates the reliability of its Key/Value contribution to the action tokens. We propose \\textbf{NoiseGate}, which combines independent per-latent timestep sampling during backbone training, a lightweight Gating Policy Network that emits per-latent time increments during denoising, and task-reward optimization that trains the schedule policy without hand-crafted shape priors. Built on a joint video--action MoT backbone, NoiseGate delivers consistent gains on diverse RoboTwin random-scene manipulation tasks.",
    "submittedDate": "2026-05-08",
    "primaryCategory": "Latent / Representation WAM",
    "secondaryCategories": [
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "NoiseGate",
    "arxivUrl": "https://arxiv.org/abs/2605.07794",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2605.07794",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "联合视频动作建模",
      "策略后训练与WM-RL"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "联合预测",
    "quadrant": "Q3 · Dual-system × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2605.07514",
    "title": "Is the Future Compatible? Diagnosing Dynamic Consistency in World Action Models",
    "authors": "Ruan, Bo-Kai; Hsiao, Teng-Fang; Lo, Ling; Shuai, Hong-Han",
    "affiliations": "National Yang Ming Chiao Tung University",
    "contribution": "Building on these findings, we introduce a value-free consensus strategy for test-time selection, which ranks candidate rollouts by agreement among predicted futures.",
    "abstract": "World Action Models (WAMs) enable decision-making through imagined rollouts by predicting future observations and actions. However, the reliability of these imagined futures remains under-examined: is a generated future merely visually plausible, or is it dynamically compatible with the action sequence it claims to model? In this work, we identify action-state consistency, the alignment between predicted actions and induced state transitions, as a missing reliability axis for WAMs. Through a systematic study across representative joint-prediction and inverse-dynamics models, we find that action-state consistency systematically separates successful and failed rollouts across many tasks and follows similar success-failure trends as learned value estimates. These results suggest that consistency captures decision-relevant structure beyond visual realism. We further identify background collapse as an important boundary condition, where low-dynamics failed trajectories can become deceptively consistent because static futures are easier to predict. Building on these findings, we introduce a value-free consensus strategy for test-time selection, which ranks candidate rollouts by agreement among predicted futures. This strategy improves success rates on RoboCasa and RoboTwin 2.0 without additional training or reward modeling. Taken together, our findings establish action-state consistency as both a diagnostic tool for evaluating WAM reliability and a practical signal for value-free planning.",
    "submittedDate": "2026-05-08",
    "primaryCategory": "Evaluation / Survey / Theory",
    "secondaryCategories": [],
    "bibtexKey": "DynamicConsistency",
    "arxivUrl": "https://arxiv.org/abs/2605.07514",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2605.07514",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "评估指标（Metrics）",
    "subcategories": [
      "世界动作一致性",
      "评估指标与协议",
      "理论与规划"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2605.07079",
    "title": "Learning Visual Feature-Based World Models via Residual Latent Action",
    "authors": "Zhang, Xinyu; Xu, Zhengtong; Tao, Yutian; Wang, Yeping; She, Yu; Boularias, Abdeslam",
    "affiliations": "1 Rutgers University 2 Purdue University 3 University of Wisconsin-Madison",
    "contribution": "Building on RLA, we propose *RLA World Model* (RLA-WM), which predicts RLA values via flow matching. RLA-WM outperforms both state-of-the-art feature-based and video-diffusion world models on simulation and real-world datasets, while being orders of magnitude faster than video diffusion.",
    "abstract": "World models predict future transitions from observations and actions. Existing works predominantly focus on image generation only. Visual feature-based world models, on the other hand, predict future visual features instead of raw video pixels, offering a promising alternative that is more efficient and less prone to hallucination. However, current feature-based approaches rely on direct regression, which leads to blurry or collapsed predictions in complex interactions, while generative modeling in high-dimensional feature spaces still remains challenging. In this work, we discover that a new type of latent action representation, which we refer to as *Residual Latent Action* (RLA), can be easily learned from DINO residuals. We also show that RLA is predictive, generalizable, and encodes temporal progression. Building on RLA, we propose *RLA World Model* (RLA-WM), which predicts RLA values via flow matching. RLA-WM outperforms both state-of-the-art feature-based and video-diffusion world models on simulation and real-world datasets, while being orders of magnitude faster than video diffusion. Furthermore, we develop two robot learning techniques that use RLA-WM to improve policy learning. The first one is a minimalist world action model with RLA that learns from actionless demonstration videos. The second one is the first visual RL framework trained entirely inside a world model learned from offline videos only, using a video-aligned reward and no online interactions or handcrafted rewards. Project page: https://mlzxy.github.io/rla-wm",
    "submittedDate": "2026-05-07",
    "primaryCategory": "Latent / Representation WAM",
    "secondaryCategories": [
      "Real-Time / Efficient WAM"
    ],
    "bibtexKey": "RLA",
    "arxivUrl": "https://arxiv.org/abs/2605.07079",
    "codeUrls": [
      "https://github.com/mlzxy/rla-wm"
    ],
    "projectUrl": "https://mlzxy.github.io/rla-wm/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2605.07079",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "潜空间预测与JEPA",
      "策略后训练与WM-RL"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2605.06481",
    "title": "OA-WAM: Object-Addressable World Action Model for Robust Robot Manipulation",
    "authors": "Liu, Yushan; Sun, Peibo; Li, Shoujie; Xie, Yifan; Zhang, Lingfeng; Chao, Xintao; Dong, Shiyuan; Chen, Fang; Zhang, Xiao-Ping; Ding, Wenbo",
    "affiliations": "1 Tsinghua University; 2 Shanghai Jiao Tong University; 3 Nanyang Technological University",
    "contribution": "We propose OA-WAM, an Object-Addressable World Action Model for robust robot manipulation. OA-WAM matches strong VLA and WAM baselines on LIBERO (97.8%) and SimplerEnv (79.3%), reaches state-of-the-art performance on the most relevant LIBERO-Plus geometric axes, and remains competitive on the seven-axis aggregate.",
    "abstract": "World Action Models (WAMs) enhance Vision-Language-Action policies by jointly predicting scene evolution and robot actions, but existing methods usually represent the predicted world as holistic images, video tokens, or global latents. These representations are difficult for an action decoder to address when an instruction refers to a particular object, especially under scene shifts where object identity is entangled with context. We propose OA-WAM, an Object-Addressable World Action Model for robust robot manipulation. OA-WAM decomposes each frame into N+1 slot states, with one robot slot and N object slots. Each slot contains a persistent address vector and a time-varying content vector, and is fused with text, image, proprioception, and past-action tokens in a block-causal sequence. A world head predicts next-frame slot states, while a flow-matching action head decodes a 16-step continuous action chunk in the same forward pass. Addressability is enforced by routing cross-slot attention through address-only keys and resetting the address slice at every transformer layer, separating which object to act on from what that object currently is without adding extra tokens. OA-WAM matches strong VLA and WAM baselines on LIBERO (97.8%) and SimplerEnv (79.3%), reaches state-of-the-art performance on the most relevant LIBERO-Plus geometric axes, and remains competitive on the seven-axis aggregate. A causal slot-intervention test yields a swap-binding cosine of 0.87, versus at most 0.09 for holistic baselines. These results suggest that addressable object states provide an effective interface for robust world-action modeling under scene perturbations.",
    "submittedDate": "2026-05-07",
    "primaryCategory": "3D/4D WAM",
    "secondaryCategories": [
      "Multimodal / Tactile WAM"
    ],
    "bibtexKey": "OAWAM",
    "arxivUrl": "https://arxiv.org/abs/2605.06481",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2605.06481",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "潜空间预测与JEPA",
      "泛化与动作对齐"
    ],
    "architecture": "One Model",
    "predictionParadigm": "联合预测",
    "quadrant": "Q1 · One Model × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2605.06247",
    "title": "CKT-WAM: Parameter-Efficient Context Knowledge Transfer Between World Action Models",
    "authors": "Jiang, Yuhua; Guo, Yijun; Yang, Hongbing; Lei, Guojun; Chen, Nuo; Zhang, Yinuo; Yan, Shaoqiang; Lin, Bo; Gao, Feifei; Qi, Biqing",
    "affiliations": "1 Tsinghua University; 3 Shanghai AI Laboratory",
    "contribution": "We propose \\textbf{CKT-WAM}, a parameter-efficient \\textbf{C}ontext \\textbf{K}nowledge \\textbf{T}ransfer framework that transfers teacher WAM's knowledge into a student WAM through a compact context in the text embedding space, rather than output imitation or dense hidden-state matching. Experiments show that CKT-WAM consistently improves zero-shot generalization and achieves the best overall performance on LIBERO-Plus, reaching 86.1\\% total success rate with only 1.17\\% trainable parameters, while approaching full fine-tuning performance.",
    "abstract": "World action models (WAMs) provide a powerful generative framework for embodied control, yet transferring knowledge across heterogeneous WAMs remains challenging due to mismatched latent interfaces, high adaptation cost, and the rigidity of conventional distillation objectives. We propose \\textbf{CKT-WAM}, a parameter-efficient \\textbf{C}ontext \\textbf{K}nowledge \\textbf{T}ransfer framework that transfers teacher WAM's knowledge into a student WAM through a compact context in the text embedding space, rather than output imitation or dense hidden-state matching. Specifically, CKT-WAM extracts intermediate teacher hidden states, reduces the number of tokens via compressors' learnable-query cross attention (LQCA), and transforms them through an always-on generalized adapter, a lightweight router, and sparsely activated specialized adapters. The resulting context is then appended to the student's conditioning textual embeddings, thereby injecting the transferred knowledge into the student with minimal architectural modification. Experiments show that CKT-WAM consistently improves zero-shot generalization and achieves the best overall performance on LIBERO-Plus, reaching 86.1\\% total success rate with only 1.17\\% trainable parameters, while approaching full fine-tuning performance. Beyond simulation, CKT-WAM also demonstrates strong real-world long-horizon manipulation ability, achieving the best average success rate of 83.3\\% across four multi-step and long-horizon tasks. Code is available at https://github.com/YuhuaJiang2002/CKT-WAM.",
    "submittedDate": "2026-05-07",
    "primaryCategory": "Real-Time / Efficient WAM",
    "secondaryCategories": [
      "Latent / Representation WAM"
    ],
    "bibtexKey": "CKTWAM",
    "arxivUrl": "https://arxiv.org/abs/2605.06247",
    "codeUrls": [
      "https://github.com/YuhuaJiang2002/CKT-WAM"
    ],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2605.06247",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "高效推理与实时控制",
      "泛化与动作对齐"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2605.06192",
    "title": "EA-WM: Event-Aware Generative World Model with Structured Kinematic-to-Visual Action Fields",
    "authors": "Yang, Zhaoyang; Jin, Yurun; Qi, Lizhe; Huang, Cong; Chen, Kai",
    "affiliations": "1 Fudan University; 2 Zhongguancun Academy; 3 Zhongguancun Institute of Artificial Intelligence; 4 University of Science and Technology of China",
    "contribution": "To bridge this gap, we present EA-WM, an Event-Aware Generative World Model that effectively closes the loop between kinematic control and visual perception. To fully exploit this geometrically grounded representation, we introduce event-aware bidirectional fusion blocks that modulate cross-branch attention, capturing object state changes and interaction dynamics.",
    "abstract": "Pretrained video diffusion models provide powerful spatiotemporal generative priors, making them a natural foundation for robotic world models. While recent world-action models jointly optimize future videos and actions, they predominantly treat video generation as an auxiliary representation for policy learning. Consequently, they insufficiently explore the inverse problem: leveraging action signals to guide video synthesis, thereby often failing to preserve precise robot spatial geometry and fine-grained robot-object interaction dynamics in the generated rollouts. To bridge this gap, we present EA-WM, an Event-Aware Generative World Model that effectively closes the loop between kinematic control and visual perception. Rather than injecting joint or end-effector actions as abstract, low-dimensional tokens, EA-WM projects actions and kinematic states directly into the target camera view as Structured Kinematic-to-Visual Action Fields. To fully exploit this geometrically grounded representation, we introduce event-aware bidirectional fusion blocks that modulate cross-branch attention, capturing object state changes and interaction dynamics. Evaluated on the comprehensive WorldArena benchmark, EA-WM achieves state-of-the-art performance, outperforming existing baselines by a significant margin.",
    "submittedDate": "2026-05-07",
    "primaryCategory": "3D/4D WAM",
    "secondaryCategories": [
      "Evaluation / Survey / Theory"
    ],
    "bibtexKey": "EAWM",
    "arxivUrl": "https://arxiv.org/abs/2605.06192",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2605.06192",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "神经世界模拟器",
      "三维表示与状态估计"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2605.00080",
    "title": "World Model for Robot Learning: A Comprehensive Survey",
    "authors": "Bohan Hou; Gen Li; Jindou Jia; Tuo An; Xinying Guo; Sicong Leng; Haoran Geng; Yanjie Ze; Tatsuya Harada; Philip Torr; Oier Mees; Marc Pollefeys; Zhuang Liu; Jiajun Wu; Pieter Abbeel; Jitendra Malik; Yilun Du; Jianfei Yang",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2026-04-30",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "hou2026world",
    "arxivUrl": "https://arxiv.org/abs/2605.00080",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2605.00080",
    "pdfUrl": "https://arxiv.org/pdf/2605.00080",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{hou2026world,\n  title={World model for robot learning: A comprehensive survey},\n  author={Hou, Bohan and Li, Gen and Jia, Jindou and An, Tuo and Guo, Xinying and Leng, Sicong and Geng, Haoran and Ze, Yanjie and Harada, Tatsuya and Torr, Philip and others},\n  journal={arXiv:2605.00080},\n  year={2026}\n}",
    "majorCategory": "Related resources",
    "subcategories": [
      "综述与技术资源"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2605.00078",
    "title": "Being-H0.7: A Latent World-Action Model from Egocentric Videos",
    "authors": "BeingBeyond Team (Hao Luo; Wanpeng Zhang; Yicheng Feng; Sipeng Zheng; Haiweng Xu; Chaoyi Xu; Ziheng Xi; Yuhui Fu; Zongqing Lu)",
    "affiliations": "",
    "contribution": "Being-H0.7 trains a robot policy to anticipate useful future information inside latent queries. A future-aware training branch aligns its hidden states with a deployable branch that sees only current context, avoiding visual rollout during control. Strong benchmark and real-robot results support the complete system, while missing component ablations leave the causal contribution of future alignment unresolved (E02–E09, E12, E15).",
    "abstract": "",
    "submittedDate": "2026-04-30",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv260500078",
    "arxivUrl": "https://arxiv.org/abs/2605.00078",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2605.00078",
    "pdfUrl": "https://arxiv.org/pdf/2605.00078",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "潜空间预测与JEPA"
    ],
    "architecture": null,
    "predictionParadigm": null,
    "quadrant": null,
    "classificationStatus": null
  },
  {
    "id": "2604.27792",
    "title": "Motubrain: An Advanced World Action Model for Robot Control",
    "authors": "Motubrain Team",
    "affiliations": "",
    "contribution": "Motubrain combines language, video and action streams in a unified generative robot policy. It transfers video priors into relative end-effector control, then uses asymmetric attention and asynchronous action chunks for deployment. Simulation, world-prediction and real-robot evaluations support different capabilities; the autoregressive attention specification remains internally inconsistent.",
    "abstract": "",
    "submittedDate": "2026-04-30",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv260427792",
    "arxivUrl": "https://arxiv.org/abs/2604.27792",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2604.27792",
    "pdfUrl": "https://arxiv.org/pdf/2604.27792",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "联合视频动作建模",
      "高效推理与实时控制"
    ],
    "architecture": null,
    "predictionParadigm": null,
    "quadrant": null,
    "classificationStatus": null
  },
  {
    "id": "2604.26694",
    "title": "Unified 4D World Action Modeling from Video Priors with Asynchronous Denoising",
    "authors": "Jun Guo; Qiwei Li; Peiyan Li; Zilong Chen; Nan Sun; Yifei Su; Heyun Wang; Yuan Zhang; Xinghang Li; Huaping Liu",
    "affiliations": "Tsinghua University; Xiaomi Robotics; Peking University; CASIA",
    "contribution": "X-WAM adapts a pretrained video diffusion transformer to predict robot actions, future states and multi-view RGB-D observations together. A one-way depth branch adds geometric supervision, while asynchronous denoising releases actions before completing video generation. The paper reports strong simulated manipulation and reconstruction results plus a small physical earphone-packing evaluation. Its central tradeoff is useful spatial supervision without depth decoding during every action step; limited temporal context and delayed control remain unresolved.",
    "abstract": "",
    "submittedDate": "2026-04-29",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv260426694",
    "arxivUrl": "https://arxiv.org/abs/2604.26694",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2604.26694",
    "pdfUrl": "https://arxiv.org/pdf/2604.26694",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "联合视频动作建模",
      "三维多视角建模"
    ],
    "architecture": "One Model",
    "predictionParadigm": "联合预测",
    "quadrant": "Q1 · One Model × 联合预测",
    "classificationStatus": null
  },
  {
    "id": "2604.25859",
    "title": "Privileged Foresight Distillation: Zero-Cost Future Correction for World Action Models",
    "authors": "Pengcheng Fang; Hongli Chen; Xiaohao Cai",
    "affiliations": "The University of Southampton; The University of Queensland",
    "contribution": "Privileged Foresight Distillation (PFD) compares two attention masks on the same world-action backbone, then learns the future-enabled change in action velocity through an output adapter. Deployment uses only the current frame and corrected action denoising. The reported LIBERO mean improves by 1.15 percentage points over reproduced Fast-WAM, with a measured 5.15% cached-context latency overhead; the title's zero-cost wording does not mean zero additional computation.",
    "abstract": "",
    "submittedDate": "2026-04-28",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv260425859",
    "arxivUrl": "https://arxiv.org/abs/2604.25859",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2604.25859",
    "pdfUrl": "https://arxiv.org/pdf/2604.25859",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "高效推理与实时控制"
    ],
    "architecture": null,
    "predictionParadigm": null,
    "quadrant": null,
    "classificationStatus": null
  },
  {
    "id": "2604.23073",
    "title": "RL Token: Bootstrapping Online RL with Vision-Language-Action Models",
    "authors": "Charles Xu; Jost Tobias Springenberg; Michael Equi; Ali Amin; Adnan Esmail; Sergey Levine; Liyiming Ke",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2026-04-24",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "rl_token",
    "arxivUrl": "https://arxiv.org/abs/2604.23073",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2604.23073",
    "pdfUrl": "https://arxiv.org/pdf/2604.23073",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{rl_token,\n  title={RL Token: Bootstrapping Online RL with Vision-Language-Action Models},\n  author={Xu, Charles and Springenberg, Jost Tobias and Equi, Michael and Amin, Ali and Esmail, Adnan and Levine, Sergey and Ke, Liyiming},\n  journal={arXiv preprint arXiv:2604.23073},\n  year={2026}\n}",
    "majorCategory": "VLA",
    "subcategories": [
      "VLA后训练与数据增强"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2604.19092",
    "title": "RoboWM-Bench: A Benchmark for Evaluating World Models in Robotic Manipulation",
    "authors": "Feng Jiang; Yang Chen; Kyle Xu; Yuchen Liu; Haifeng Wang; Zhenhao Shen; Jasper Lu; Shengze Huang; Yuanfei Wang; Chen Xie; Ruihai Wu",
    "affiliations": "Peking University; Tsinghua University; Lightwheel",
    "contribution": "RoboWM-Bench evaluates whether generated manipulation videos can be converted into robot actions that complete tasks in simulation. Separate human-hand retargeting and robot inverse-dynamics interfaces expose failures hidden by plausible imagery. Reliability tests support these interfaces on real demonstrations, but execution scores still depend on action extraction, reconstructed physics and task-specific checkers.",
    "abstract": "",
    "submittedDate": "2026-04-21",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv260419092",
    "arxivUrl": "https://arxiv.org/abs/2604.19092",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2604.19092",
    "pdfUrl": "https://arxiv.org/pdf/2604.19092",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "世界模型评测基准",
      "机器人操作基准"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": null
  },
  {
    "id": "2604.15483",
    "title": "π_0.7: a Steerable Generalist Robotic Foundation Model with Emergent Capabilities",
    "authors": "Physical Intelligence; Bo Ai; Ali Amin; Raichelle Aniceto; Ashwin Balakrishna; Greg Balke; Kevin Black; George Bokinsky; Shihao Cao; Thomas Charbonnier; Vedant Choudhary; Foster Collins; Ken Conley; Grace Connors; James Darpinian; Karan Dhabalia; Maitrayee Dhaka; Jared DiCarlo; Danny Driess; Michael Equi; Adnan Esmail; Yunhao Fang; Chelsea Finn; Catherine Glossop; Thomas Godden; Ivan Goryachev; Lachlan Groom; Haroun Habeeb; Hunter Hancock; Karol Hausman; Gashon Hussein; Victor Hwang; Brian Ichter; Connor Jacobsen; Szymon Jakubczak; Rowan Jen; Tim Jones; Gregg Kammerer; Ben Katz; Liyiming Ke; Mairbek Khadikov; Chandra Kuchi; Marinda Lamb; Devin LeBlanc; Brendon LeCount; Sergey Levine; Xinyu Li; Adrian Li-Bell; Vladislav Lialin; Zhonglin Liang; Wallace Lim; Yao Lu; Enyu Luo; Vishnu Mano; Nandan Marwaha; Aikys Mongush; Liam Murphy; Suraj Nair; Tyler Patterson; Karl Pertsch; Allen Z. Ren; Gavin Schelske; Charvi Sharma; Baifeng Shi; Lucy Xiaoyang Shi; Laura Smith; Jost Tobias Springenberg; Kyle Stachowicz; Will Stoeckle; Jiaming Tang; Jimmy Tanner; Shalom Tekeste; Marcel Torne; Kyle Vedder; Quan Vuong; Anna Walling; Haohuan Wang; Jason Wang; XuDong Wang; Chris Whalen; Samuel Whitmore; Blake Williams; Charles Xu; Sukwon Yoo; Lili Yu; Wuming Zhang; Zhuoyang Zhang; Ury Zhilinsky",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2026-04-16",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "pi0_7",
    "arxivUrl": "https://arxiv.org/abs/2604.15483",
    "codeUrls": [],
    "projectUrl": "https://www.pi.website/blog/pi07",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2604.15483",
    "pdfUrl": "https://arxiv.org/pdf/2604.15483",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{pi0_7,\n      title={{${\\pi}_{0.7}$: a Steerable Generalist Robotic Foundation Model with Emergent Capabilities}}, \n      author={Physical Intelligence and Bo Ai and Ali Amin and Raichelle Aniceto and Ashwin Balakrishna and Greg Balke and Kevin Black and George Bokinsky and Shihao Cao and Thomas Charbonnier and Vedant Choudhary and Foster Collins and Ken Conley and Grace Connors and James Darpinian and Karan Dhabalia and Maitrayee Dhaka and Jared DiCarlo and Danny Driess and Michael Equi and Adnan Esmail and Yunhao Fang and Chelsea Finn and Catherine Glossop and Thomas Godden and Ivan Goryachev and Lachlan Groom and Haroun Habeeb and Hunter Hancock and Karol Hausman and Gashon Hussein and Victor Hwang and Brian Ichter and Connor Jacobsen and Szymon Jakubczak and Rowan Jen and Tim Jones and Gregg Kammerer and Ben Katz and Liyiming Ke and Mairbek Khadikov and Chandra Kuchi and Marinda Lamb and Devin LeBlanc and Brendon LeCount and Sergey Levine and Xinyu Li and Adrian Li-Bell and Vladislav Lialin and Zhonglin Liang and Wallace Lim and Yao Lu and Enyu Luo and Vishnu Mano and Nandan Marwaha and Aikys Mongush and Liam Murphy and Suraj Nair and Tyler Patterson and Karl Pertsch and Allen Z. Ren and Gavin Schelske and Charvi Sharma and Baifeng Shi and Lucy Xiaoyang Shi and Laura Smith and Jost Tobias Springenberg and Kyle Stachowicz and Will Stoeckle and Jiaming Tang and Jimmy Tanner and Shalom Tekeste and Marcel Torne and Kyle Vedder and Quan Vuong and Anna Walling and Haohuan Wang and Jason Wang and XuDong Wang and Chris Whalen and Samuel Whitmore and Blake Williams and Charles Xu and Sukwon Yoo and Lili Yu and Wuming Zhang and Zhuoyang Zhang and Ury Zhilinsky},\n      year={2026},\n      journal={arXiv:2604.15483},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "视觉规划与IDM",
      "泛化与动作对齐"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2604.16484",
    "title": "DexWorldModel: Causal Latent World Modeling towards Automated Learning of Embodied Tasks",
    "authors": "Yueci Deng; Guiliang Liu; Kui Jia",
    "affiliations": "DexForce AI",
    "contribution": "DexWorldModel introduces CLWM, which predicts future DINOv3 features and then generates actions conditioned on that prediction. Shared transformer blocks, separate persistent and speculative memories, and asynchronous denoising connect world prediction to robot execution. EmbodiChain supplies synthetic adaptation data. Reported manipulation results are strong, but protocol omissions and a contradictory flow-time convention limit reproducibility.",
    "abstract": "",
    "submittedDate": "2026-04-13",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv260416484",
    "arxivUrl": "https://arxiv.org/abs/2604.16484",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2604.16484",
    "pdfUrl": "https://arxiv.org/pdf/2604.16484",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "视觉规划与IDM",
      "潜空间预测与JEPA"
    ],
    "architecture": "One Model",
    "predictionParadigm": "IDM",
    "quadrant": "Q2 · One Model × IDM",
    "classificationStatus": null
  },
  {
    "id": "2604.11135",
    "title": "AIM: Intent-Aware Unified world action Modeling with Spatial Value Maps",
    "authors": "Liaoyuan Fan; Zetian Xu; Chen Cao; Wenyao Zhang; Mingqi Yuan; Jiayu Chen",
    "affiliations": "INFIFORCE Intelligent Technology Co., Ltd., Hangzhou, China; The University of Hong Kong, Hong Kong SAR, China; Shanghai Jiao Tong University, Shanghai, China",
    "contribution": "AIM jointly generates future RGB observations, spatial contact maps and continuous robot actions. A masked mixture-of-transformers architecture makes the predicted maps the action head's only route to future information. Supervised learning uses simulator-derived contact labels; post-training freezes world/value prediction and refines actions using map responses plus task rewards. The paper reports 94.0% Easy and 92.1% Hard success on RoboTwin 2.0, but the small post-training improvement, missing mechanism ablations and inconsistent schematic require careful interpretation.",
    "abstract": "",
    "submittedDate": "2026-04-13",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv260411135",
    "arxivUrl": "https://arxiv.org/abs/2604.11135",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2604.11135",
    "pdfUrl": "https://arxiv.org/pdf/2604.11135",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "联合视频动作建模",
      "策略后训练与WM-RL"
    ],
    "architecture": null,
    "predictionParadigm": null,
    "quadrant": null,
    "classificationStatus": null
  },
  {
    "id": "2604.09330",
    "title": "VAG: Dual-Stream Video-Action Generation for Embodied Data Synthesis",
    "authors": "Xiaolei Lang; Yang Wang; Yukun Zhou; Chaojun Ni; Kerui Li; Jiagang Zhu; Tianze Liu; Jiajun Lv; Xingxing Zuo; Yun Ye; Guan Huang; Xiaofeng Wang; Zheng Zhu",
    "affiliations": "GigaAI; Zhejiang University; Peking University; Institute of Automation, Chinese Academy of Sciences; Robotics Department, Mohamed bin Zayed University of Artificial Intelligence",
    "contribution": "VAG synthesizes robot videos and action sequences from an initial image and instruction. A video diffusion transformer supplies pooled clean-latent predictions to an action U-Net during synchronized denoising. The evidence supports improved action prediction and simulation replay, plus a small real-robot policy-pretraining benefit. It does not establish guaranteed alignment or general closed-loop control.",
    "abstract": "",
    "submittedDate": "2026-04-10",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv260409330",
    "arxivUrl": "https://arxiv.org/abs/2604.09330",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2604.09330",
    "pdfUrl": "https://arxiv.org/pdf/2604.09330",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "联合视频动作建模"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "联合预测",
    "quadrant": "Q3 · Dual-system × 联合预测",
    "classificationStatus": null
  },
  {
    "id": "2604.09059",
    "title": "Learning Vision-Language-Action World Models for Autonomous Driving",
    "authors": "Guoqing Wang; Pin Tang; Xiangxuan Ren; Guodongfang Zhao; Bailan Feng; Chao Ma",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2026-04-10",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "vla_wm",
    "arxivUrl": "https://arxiv.org/abs/2604.09059",
    "codeUrls": [],
    "projectUrl": "https://vlaworld.github.io",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2604.09059",
    "pdfUrl": "https://arxiv.org/pdf/2604.09059",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{vla_wm,\n      title={Learning Vision-Language-Action World Models for Autonomous Driving}, \n      author={Guoqing Wang and Pin Tang and Xiangxuan Ren and Guodongfang Zhao and Bailan Feng and Chao Ma},\n      year={2026},\n      journal={arXiv:2604.09059},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "自动驾驶",
      "视觉规划与IDM"
    ],
    "architecture": "One Model",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2604.08719",
    "title": "LMGenDrive: Bridging Multimodal Understanding and Generative World Modeling for End-to-End Driving",
    "authors": "Hao Shao; Letian Wang; Yang Zhou; Yuxuan Hu; Zhuofan Zong; Steven L. Waslander; Wei Zhan; Hongsheng Li",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2026-04-09",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "lmgendrive",
    "arxivUrl": "https://arxiv.org/abs/2604.08719",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2604.08719",
    "pdfUrl": "https://arxiv.org/pdf/2604.08719",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{lmgendrive,\n      title={LMGenDrive: Bridging Multimodal Understanding and Generative World Modeling for End-to-End Driving}, \n      author={Hao Shao and Letian Wang and Yang Zhou and Yuxuan Hu and Zhuofan Zong and Steven L. Waslander and Wei Zhan and Hongsheng Li},\n      year={2026},\n      journal={arXiv:2604.08719},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "自动驾驶",
      "联合视频动作建模"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "联合预测",
    "quadrant": "Q3 · Dual-system × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2604.08534",
    "title": "ActiveGlasses: Learning Manipulation with Active Vision from Ego-centric Human Demonstration",
    "authors": "Yanwen Zou; Chenyang Shi; Wenye Yu; Han Xue; Jun Lv; Ye Pan; Chuan Wen; Cewu Lu",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2026-04-09",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "activeglasses",
    "arxivUrl": "https://arxiv.org/abs/2604.08534",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2604.08534",
    "pdfUrl": "https://arxiv.org/pdf/2604.08534",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{activeglasses,\n      title={ActiveGlasses: Learning Manipulation with Active Vision from Ego-centric Human Demonstration}, \n      author={Yanwen Zou and Chenyang Shi and Wenye Yu and Han Xue and Jun Lv and Ye Pan and Chuan Wen and Cewu Lu},\n      year={2026},\n      journal={arXiv:2604.08534},\n}",
    "majorCategory": "数据集",
    "subcategories": [
      "数据集与数据采集",
      "人类第一视角数据",
      "数据采集接口"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2604.07209",
    "title": "INSPATIO-WORLD: A Real-Time 4D World Simulator via Spatiotemporal Autoregressive Modeling",
    "authors": "InSpatio Team; Donghui Shen; Guofeng Zhang; Haomin Liu; Haoyu Ji; Hujun Bao; Hongjia Zhai; Jialin Liu; Jing Guo; Nan Wang; Siji Pan; Weihong Pan; Weijian Xie; Xianbin Liu; Xiaojun Xiang; Xiaoyu Zhang; Xinyu Chen; Yifu Wang; Yipeng Chen; Zhenzhou Fan; Zhewen Le; Zhichao Ye; Ziqiang Zhao",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2026-04-08",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "inspatio_world",
    "arxivUrl": "https://arxiv.org/abs/2604.07209",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2604.07209",
    "pdfUrl": "https://arxiv.org/pdf/2604.07209",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{inspatio_world,\n      title={Inspatio-world: A real-time 4d world simulator via spatiotemporal autoregressive modeling}, \n      author={Team, InSpatio and Shen, Donghui and Zhang, Guofeng and Liu, Haomin and Ji, Haoyu and Bao, Hujun and Zhai, Hongjia and Liu, Jialin and Guo, Jing and Wang, Nan and others},\n      year={2026},\n      journal={arXiv:2604.07209},\n}",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "神经世界模拟器"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2604.06168",
    "title": "Action Images: End-to-End Policy Learning via Multiview Video Generation",
    "authors": "Haoyu Zhen; Zixian Gao; Qiao Sun; Yilin Zhao; Yuncong Yang; Yilun Du; Pengsheng Guo; Tsun-Hsuan Wang; Yi-Ling Qiao; Chuang Gan",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2026-04-07",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "action_images; MultiviewVideo",
    "arxivUrl": "https://arxiv.org/abs/2604.06168",
    "codeUrls": [
      "https://github.com/UMass-Embodied-AGI/ActionImages"
    ],
    "projectUrl": "https://actionimages.github.io/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2604.06168",
    "pdfUrl": "https://arxiv.org/pdf/2604.06168",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{action_images,\n      title={Action Images: End-to-End Policy Learning via Multiview Video Generation}, \n      author={Haoyu Zhen and Zixian Gao and Qiao Sun and Yilin Zhao and Yuncong Yang and Yilun Du and Pengsheng Guo and Tsun-Hsuan Wang and Yi-Ling Qiao and Chuang Gan},\n      year={2026},\n      journal={arXiv:2604.06168},\n}\n\n@article{MultiviewVideo,\n      title={Action Images: End-to-End Policy Learning via Multiview Video Generation}, \n      author={Haoyu Zhen and Zixian Gao and Qiao Sun and Yilin Zhao and Yuncong Yang and Yilun Du and Pengsheng Guo and Tsun-Hsuan Wang and Yi-Ling Qiao and Chuang Gan},\n      year={2026},\n      journal={arXiv:2604.06168},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "联合视频动作建模",
      "三维多视角建模"
    ],
    "architecture": "One Model",
    "predictionParadigm": "联合预测",
    "quadrant": "Q1 · One Model × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2604.05498",
    "title": "JailWAM: Jailbreaking World Action Models in Robot Control",
    "authors": "Hanqing Liu; Songping Wang; Jiahuan Long; Jiacheng Hou; Jialiang Sun; Chao Li; Yang Yang; Wei Peng; Xu Liu; Tingsong Jiang; Yao Mu; Wen Yao",
    "affiliations": "MoE Key Lab of Artificial Intelligence, AI Institute, Shanghai Jiao Tong University, Shanghai, China; PR Lab, Nanjing University, Suzhou, China; Defense Innovation Institute, Chinese Academy of Military Science, Beijing, China",
    "contribution": "JailWAM evaluates instruction-induced robot risk by rendering predicted actions as trajectory charts, screening them with a trained vision-language discriminator, and verifying selected candidates in closed-loop simulation. It reports 84.20% human-verified attack success on LingBot-VA, but its cheaper screening pipeline misses some unsafe executions. Success includes motion failure as well as catastrophic risk.",
    "abstract": "",
    "submittedDate": "2026-04-07",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv260405498",
    "arxivUrl": "https://arxiv.org/abs/2604.05498",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2604.05498",
    "pdfUrl": "https://arxiv.org/pdf/2604.05498",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "评测协议与诊断",
      "鲁棒性与泛化评测"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": null
  },
  {
    "id": "2604.04198",
    "title": "DriveVA: Video Action Models are Zero-Shot Drivers",
    "authors": "Mengmeng Liu; Diankun Zhang; Jiuming Liu; Jianfeng Cui; Hongwei Xie; Guang Chen; Hangjun Ye; Michael Ying Yang; Francesco Nex; Hao Cheng",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2026-04-05",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "driveva",
    "arxivUrl": "https://arxiv.org/abs/2604.04198",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2604.04198",
    "pdfUrl": "https://arxiv.org/pdf/2604.04198",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{driveva,\n      title={DriveVA: Video Action Models are Zero-Shot Drivers}, \n      author={Mengmeng Liu and Diankun Zhang and Jiuming Liu and Jianfeng Cui and Hongwei Xie and Guang Chen and Hangjun Ye and Michael Ying Yang and Francesco Nex and Hao Cheng},\n      year={2026},\n      journal={arXiv:2604.04198},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "自动驾驶",
      "联合视频动作建模",
      "泛化与动作对齐"
    ],
    "architecture": "One Model",
    "predictionParadigm": "联合预测",
    "quadrant": "Q1 · One Model × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2604.03181",
    "title": "SpatialVAM:Spatial-Aware Multi-View Video Diffusion as a Data-Efficient Robot Policy",
    "authors": "Peiyan Li; Yixiang Chen; Yuan Xu; Jiabing Yang; Xiangnan Wu; Jun Guo; Nan Sun; Long Qian; Xinghang Li; Xin Xiao; Jing Liu; Nianfeng Liu; Tao Kong; Yan Huang; Liang Wang; Tieniu Tan",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2026-04-03",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "mvdp",
    "arxivUrl": "https://arxiv.org/abs/2604.03181",
    "codeUrls": [],
    "projectUrl": "https://spatialvam.github.io/home_page.html",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2604.03181",
    "pdfUrl": "https://arxiv.org/pdf/2604.03181",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{mvdp,\n      title={Multi-View Video Diffusion Policy: A 3D Spatio-Temporal-Aware Video Action Model}, \n      author={Li, Peiyan and Chen, Yixiang and Xu, Yuan and Yang, Jiabing and Wu, Xiangnan and Guo, Jun and Sun, Nan and Qian, Long and Li, Xinghang and Xiao, Xin and others},\n      year={2026},\n      journal={arXiv:2604.03181},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "联合视频动作建模",
      "三维多视角建模"
    ],
    "architecture": "One Model",
    "predictionParadigm": "联合预测",
    "quadrant": "Q1 · One Model × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2604.02190",
    "title": "UniDriveVLA: Unifying Understanding, Perception, and Action Planning for Autonomous Driving",
    "authors": "Yongkang Li; Lijun Zhou; Sixu Yan; Bencheng Liao; Tianyi Yan; Kaixin Xiong; Long Chen; Hongwei Xie; Bing Wang; Guang Chen; Hangjun Ye; Wenyu Liu; Haiyang Sun; Xinggang Wang",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2026-04-02",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "unidrivevla",
    "arxivUrl": "https://arxiv.org/abs/2604.02190",
    "codeUrls": [
      "https://github.com/xiaomi-research/unidrivevla"
    ],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2604.02190",
    "pdfUrl": "https://arxiv.org/pdf/2604.02190",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{unidrivevla,\n      title={UniDriveVLA: Unifying Understanding, Perception, and Action Planning for Autonomous Driving}, \n      author={Yongkang Li and Lijun Zhou and Sixu Yan and Bencheng Liao and Tianyi Yan and Kaixin Xiong and Long Chen and Hongwei Xie and Bing Wang and Guang Chen and Hangjun Ye and Wenyu Liu and Haiyang Sun and Xinggang Wang},\n      year={2026},\n      journal={arXiv:2604.02190},\n}",
    "majorCategory": "VLA",
    "subcategories": [
      "分层与双系统VLA",
      "自动驾驶VLA",
      "空间感知增强VLA"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2604.01765",
    "title": "DriveDreamer-Policy: A Geometry-Grounded World-Action Model for Unified Generation and Planning",
    "authors": "Yang Zhou; Xiaofeng Wang; Hao Shao; Letian Wang; Guosheng Zhao; Jiangnan Shao; Jiagang Zhu; Tingdong Yu; Zheng Zhu; Guan Huang; Steven L. Waslander",
    "affiliations": "GigaAI; University of Toronto; CUHK MMLab",
    "contribution": "DriveDreamer-Policy trains a shared multimodal backbone with depth, video and trajectory generators. Ordered query embeddings transfer geometry and future-scene context into planning without requiring rendered depth or video at inference. Navsim scores and controlled modality ablations support useful joint supervision, while depth evaluation against a learned teacher and incomplete runtime details limit the conclusions.",
    "abstract": "",
    "submittedDate": "2026-04-02",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv260401765",
    "arxivUrl": "https://arxiv.org/abs/2604.01765",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2604.01765",
    "pdfUrl": "https://arxiv.org/pdf/2604.01765",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "自动驾驶",
      "联合视频动作建模"
    ],
    "architecture": null,
    "predictionParadigm": null,
    "quadrant": null,
    "classificationStatus": null
  },
  {
    "id": "2603.28955",
    "title": "Enhancing Policy Learning with World-Action Model",
    "authors": "Yuci Han; Alper Yilmaz",
    "affiliations": "Photogrammetry and Computer Vision Lab, The Ohio State University, Columbus, OH 43210, USA",
    "contribution": "WAM adds inverse action prediction between consecutive encoder embeddings to a DreamerV2 world model, then trains a separate diffusion policy on its frozen latent features. Table III supports 61.7% versus 45.8% average behavioral-cloning success on eight CALVIN tasks. The mechanism is plausible, but inconsistent headline numbers, baseline names and PPO summaries limit stronger conclusions.",
    "abstract": "",
    "submittedDate": "2026-03-30",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv260328955",
    "arxivUrl": "https://arxiv.org/abs/2603.28955",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2603.28955",
    "pdfUrl": "https://arxiv.org/pdf/2603.28955",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "策略后训练与WM-RL"
    ],
    "architecture": "待核实",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": null
  },
  {
    "id": "2603.28545",
    "title": "ManipArena: A Controlled Benchmark for Diagnosing Generalization in Real-Robot Manipulation",
    "authors": "Yu Sun; Meng Cao; Yang Ping; Kaidong Zhang; Qingxuan Chen; Rongtao Xu; Liangwang Ruan; Xuecheng Chen; Dongxiu Liu; Yunxiao Yan; Zunnan Xu; Runze Xu; Charles Yang; Peilun Zhang; Xiaofan Li; Ruyi Gan; Liang Ma; Yuehao Yin; Jincheng Yu; Lufang Chen; Yuxin Liang; Peng Zhai; Hao Wang; Ivan Laptev; Ian Reid; Qian Wang; Xiaodan Liang",
    "affiliations": "Sun Yat-sen University; X Square Robot; MBZUAI; Tsinghua University; University of Zurich",
    "contribution": "ManipArena evaluates manipulation policies through controlled physical robot trials, task schemas and subgoal scoring. Its strongest lesson is that training recipes and model provenance affect rankings alongside architecture. Language grounding and demonstration selection produce substantial reported gains, but small trial counts, restricted environments and internal reporting inconsistencies limit causal and generalization claims.",
    "abstract": "",
    "submittedDate": "2026-03-30",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv260328545",
    "arxivUrl": "https://arxiv.org/abs/2603.28545",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2603.28545",
    "pdfUrl": "https://arxiv.org/pdf/2603.28545",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "机器人操作基准",
      "泛化评测"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": null
  },
  {
    "id": "2603.27287",
    "title": "Uni-World VLA: Interleaved World Modeling and Planning for Autonomous Driving",
    "authors": "Qiqi Liu; Huan Xu; Jingyu Li; Bin Sun; Zhihui Hao; Dangen She; Xiatian Zhu; Li Zhang",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2026-03-28",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "uni_world_vla",
    "arxivUrl": "https://arxiv.org/abs/2603.27287",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2603.27287",
    "pdfUrl": "https://arxiv.org/pdf/2603.27287",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{uni_world_vla,\n      title={Uni-World VLA: Interleaved World Modeling and Planning for Autonomous Driving}, \n      author={Qiqi Liu and Huan Xu and Jingyu Li and Bin Sun and Zhihui Hao and Dangen She and Xiatian Zhu and Li Zhang},\n      year={2026},\n      journal={arXiv:2603.27287},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "视觉规划与IDM",
      "自动驾驶"
    ],
    "architecture": "One Model",
    "predictionParadigm": "IDM",
    "quadrant": "Q2 · One Model × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2603.25741",
    "title": "Vega: Learning to Drive with Natural Language Instructions",
    "authors": "Sicheng Zuo; Yuxuan Li; Wenzhao Zheng; Zheng Zhu; Jie Zhou; Jiwen Lu",
    "affiliations": "Tsinghua University; GigaAI",
    "contribution": "Vega learns instruction-conditioned driving by training trajectory denoising together with future-image denoising. Modality-specific transformers exchange information through global causal attention. InstructScene supplies automatically generated instructions describing recorded driving. The strongest reported NAVSIM v2 score uses best-of-six trajectory selection; NAVSIM v1 results are weaker than leading VLA baselines. Future-prediction ablations support visual supervision, while selected images illustrate instruction sensitivity without measuring general instruction-following reliability.",
    "abstract": "",
    "submittedDate": "2026-03-26",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv260325741",
    "arxivUrl": "https://arxiv.org/abs/2603.25741",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2603.25741",
    "pdfUrl": "https://arxiv.org/pdf/2603.25741",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "自动驾驶",
      "联合视频动作建模"
    ],
    "architecture": null,
    "predictionParadigm": null,
    "quadrant": null,
    "classificationStatus": null
  },
  {
    "id": "2603.25685",
    "title": "Persistent Robot World Models: Stabilizing Multi-Step Rollouts via Reinforcement Learning",
    "authors": "Jai Bardhan; Patrik Drozdik; Josef Sivic; Vladimir Petrik",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2026-03-26",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "prwm",
    "arxivUrl": "https://arxiv.org/abs/2603.25685",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2603.25685",
    "pdfUrl": "https://arxiv.org/pdf/2603.25685",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{prwm,\n      title={Persistent Robot World Models: Stabilizing Multi-Step Rollouts via Reinforcement Learning}, \n      author={Jai Bardhan and Patrik Drozdik and Josef Sivic and Vladimir Petrik},\n      year={2026},\n      journal={arXiv:2603.25685},\n}",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "神经世界模拟器",
      "训练优化与蒸馏"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2603.24587",
    "title": "DreamerAD: Efficient Reinforcement Learning via Latent World Model for Autonomous Driving",
    "authors": "Pengxuan Yang; Yupeng Zheng; Deheng Qian; Zebin Xing; Qichao Zhang; Linbo Wang; Yichen Zhang; Shaoyu Guo; Zhongpu Xia; Qiang Chen; Junyu Han; Lingyun Xu; Yifeng Pan; Dongbin Zhao",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2026-03-25",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "dreamerad",
    "arxivUrl": "https://arxiv.org/abs/2603.24587",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2603.24587",
    "pdfUrl": "https://arxiv.org/pdf/2603.24587",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{dreamerad,\n      title={DreamerAD: Efficient Reinforcement Learning via Latent World Model for Autonomous Driving}, \n      author={Pengxuan Yang and Yupeng Zheng and Deheng Qian and Zebin Xing and Qichao Zhang and Linbo Wang and Yichen Zhang and Shaoyu Guo and Zhongpu Xia and Qiang Chen and Junyu Han and Lingyun Xu and Yifeng Pan and Dongbin Zhao},\n      year={2026},\n      journal={arXiv:2603.24587},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "策略后训练与WM-RL",
      "自动驾驶",
      "高效推理与实时控制"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2603.24581",
    "title": "Latent-WAM: Latent World Action Modeling for End-to-End Autonomous Driving",
    "authors": "Linbo Wang; Yupeng Zheng; Qiang Chen; Shiwei Li; Yichen Zhang; Zebin Xing; Qichao Zhang; Xiang Li; Deheng Qian; Pengxuan Yang; Yihang Dong; Ce Hao; Xiaoqing Ye; Junyu han; Yifeng Pan; Dongbin Zhao",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2026-03-25",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "latent_wam",
    "arxivUrl": "https://arxiv.org/abs/2603.24581",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2603.24581",
    "pdfUrl": "https://arxiv.org/pdf/2603.24581",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{latent_wam,\n      title={Latent-WAM: Latent World Action Modeling for End-to-End Autonomous Driving}, \n      author={Linbo Wang and Yupeng Zheng and Qiang Chen and Shiwei Li and Yichen Zhang and Zebin Xing and Qichao Zhang and Xiang Li and Deheng Qian and Pengxuan Yang and Yihang Dong and Ce Hao and Xiaoqing Ye and Junyu han and Yifeng Pan and Dongbin Zhao},\n      year={2026},\n      journal={arXiv:2603.24581},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "自动驾驶",
      "潜空间预测与JEPA"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2603.24506",
    "title": "Toward Physically Consistent Driving Video World Models under Challenging Trajectories",
    "authors": "Jiawei Zhou; Zhenxin Zhu; Lingyi Du; Linye Lyu; Lijun Zhou; Zhanqian Wu; Hongcheng Luo; Zhuotao Tian; Bing Wang; Guang Chen; Hangjun Ye; Haiyang Sun; Yu Li",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2026-03-25",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "phygenesis",
    "arxivUrl": "https://arxiv.org/abs/2603.24506",
    "codeUrls": [],
    "projectUrl": "https://wm-research.github.io/PhyGenesis/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2603.24506",
    "pdfUrl": "https://arxiv.org/pdf/2603.24506",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{phygenesis,\n      title={Toward Physically Consistent Driving Video World Models under Challenging Trajectories}, \n      author={Jiawei Zhou and Zhenxin Zhu and Lingyi Du and Linye Lyu and Lijun Zhou and Zhanqian Wu and Hongcheng Luo and Zhuotao Tian and Bing Wang and Guang Chen and Hangjun Ye and Haiyang Sun and Yu Li},\n      year={2026},\n      journal={arXiv:2603.24506},\n}",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "神经世界模拟器",
      "自动驾驶"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2603.23481",
    "title": "VTAM: Video-Tactile-Action Models for Complex Physical Interaction Beyond VLAs",
    "authors": "Haoran Yuan; Weigang Yi; Zhenyu Zhang; Wendi Chen; Yuchen Mo; Jiashi Yin; Xinzhuo Li; Xiangyu Zeng; Chuan Wen; Cewu Lu; Katherine Driggs-Campbell; Ismini Lourentzou",
    "affiliations": "University of Illinois Urbana-Champaign; Stanford University; Shanghai Jiao Tong University",
    "contribution": "VTAM adapts a video world model to predict camera and tactile streams, then trains a conditional action expert with an auxiliary deformation-derived force target. It reports large gains in three real robot contact tasks. The strongest evidence is task success and a chip ablation; calibrated force accuracy, broad generalization and the claimed gradient mechanism remain unestablished. Method details below preserve inconsistencies between the formulation and implementation appendix.",
    "abstract": "",
    "submittedDate": "2026-03-24",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv260323481",
    "arxivUrl": "https://arxiv.org/abs/2603.23481",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2603.23481",
    "pdfUrl": "https://arxiv.org/pdf/2603.23481",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "多模态触觉音频"
    ],
    "architecture": null,
    "predictionParadigm": null,
    "quadrant": null,
    "classificationStatus": null
  },
  {
    "id": "2603.22078",
    "title": "Do World Action Models Generalize Better than VLAs? A Robustness Study",
    "authors": "Zhanguang Zhang; Zhiyuan Li; Behnam Rahmati; Rui Heng Yang; Yintao Ma; Amir Rasouli; Sajjad Pakdamansavoji; Yangzheng Wu; Lingfeng Zhang; Tongtong Cao; Feng Wen; Xinyu Wang; Xingyue Quan; Yingxue Zhang",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2026-03-23",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "dwam",
    "arxivUrl": "https://arxiv.org/abs/2603.22078",
    "codeUrls": [
      "https://github.com/Robot-Robustness/RoboTwin2.0-Plus"
    ],
    "projectUrl": "https://robot-robustness.github.io/RoboTwin2.0-Plus/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2603.22078",
    "pdfUrl": "https://arxiv.org/pdf/2603.22078",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{dwam,\n      title={Do World Action Models Generalize Better than VLAs? A Robustness Study}, \n      author={Zhanguang Zhang and Zhiyuan Li and Behnam Rahmati and Rui Heng Yang and Yintao Ma and Amir Rasouli and Sajjad Pakdamansavoji and Yangzheng Wu and Lingfeng Zhang and Tongtong Cao and Feng Wen and Xinyu Wang and Xingyue Quan and Yingxue Zhang},\n      year={2026},\n      journal={arXiv:2603.22078},\n}",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "机器人操作基准",
      "鲁棒性与泛化评测",
      "评测协议与诊断"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2603.19979",
    "title": "X-World: Controllable Ego-Centric Multi-Camera World Models for Scalable End-to-End Driving",
    "authors": "Chaoda Zheng; Sean Li; Jinhao Deng; Zhennan Wang; Shijia Chen; Liqiang Xiao; Ziheng Chi; Hongbin Lin; Kangjie Chen; Boyang Wang; Yu Zhang; Xianming Liu",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2026-03-20",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "x_world",
    "arxivUrl": "https://arxiv.org/abs/2603.19979",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2603.19979",
    "pdfUrl": "https://arxiv.org/pdf/2603.19979",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{x_world,\n      title={X-World: Controllable Ego-Centric Multi-Camera World Models for Scalable End-to-End Driving}, \n      author={Zheng, Chaoda and Li, Sean and Deng, Jinhao and Wang, Zhennan and Chen, Shijia and Xiao, Liqiang and Chi, Ziheng and Lin, Hongbin and Chen, Kangjie and Wang, Boyang and others},\n      year={2026},\n      journal={arXiv:2603.19979},\n}",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "神经世界模拟器"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2603.19370",
    "title": "VAMPO: Policy Optimization for Improving Visual Dynamics in Video Action Models",
    "authors": "Zirui Ge; Pengxiang Ding; Baohua Yin; Qishen Wang; Zhiyong Xie; Yemin Wang; Jinbo Wang; Hengtao Li; Runze Suo; Wenxuan Song; Han Zhao; Shangke Lyu; Zhaoxin Fan; Haoang Li; Ran Cheng; Cheng Chi; Huibin Ge; Yaozhi Luo; Donglin Wang",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2026-03-19",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "vampo",
    "arxivUrl": "https://arxiv.org/abs/2603.19370",
    "codeUrls": [
      "https://github.com/OpenHelix-Team/VAMPO"
    ],
    "projectUrl": "https://vampo-robot.github.io/VAMPO/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2603.19370",
    "pdfUrl": "https://arxiv.org/pdf/2603.19370",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{vampo,\n      title={VAMPO: Policy Optimization for Improving Visual Dynamics in Video Action Models}, \n      author={Ge, Zirui and Ding, Pengxiang and Yin, Baohua and Wang, Qishen and Xie, Zhiyong and Wang, Yemin and Wang, Jinbo and Li, Hengtao and Suo, Runze and Song, Wenxuan and others},\n      year={2026},\n      journal={arXiv:2603.19370},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "策略后训练与WM-RL",
      "视觉规划与IDM"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2603.17808",
    "title": "EVA: Aligning Video World Models with Executable Robot Actions via Inverse Dynamics Rewards",
    "authors": "Ruixiang Wang; Qingming Liu; Yueci Deng; Guiliang Liu; Zhen Liu; Kui Jia",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2026-03-18",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "eva_vwm",
    "arxivUrl": "https://arxiv.org/abs/2603.17808",
    "codeUrls": [
      "https://github.com/RobbinW/EVA"
    ],
    "projectUrl": "https://eva-project-page.github.io/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2603.17808",
    "pdfUrl": "https://arxiv.org/pdf/2603.17808",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{eva_vwm,\n      title={EVA: Aligning Video World Models with Executable Robot Actions via Inverse Dynamics Rewards}, \n      author={Ruixiang Wang and Qingming Liu and Yueci Deng and Guiliang Liu and Zhen Liu and Kui Jia},\n      year={2026},\n      journal={arXiv:2603.17808},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "策略后训练与WM-RL",
      "视觉规划与IDM"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2603.17240",
    "title": "GigaWorld-Policy: An Efficient Action-Centered World–Action Model",
    "authors": "Angen Ye; Boyuan Wang; Chaojun Ni; Guan Huang; Guosheng Zhao; Hao Li; Hengtao Li; Jie Li; Jindi Lv; Jingyu Liu; Min Cao; Peng Li; Qiuping Deng; Wenjun Mei; Xiaofeng Wang; Xinze Chen; Xinyu Zhou; Yang Wang; Yifan Chang; Yifan Li; Yukun Zhou; Yun Ye; Zhichao Liu; Zheng Zhu",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2026-03-18",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "gigaworld_policy",
    "arxivUrl": "https://arxiv.org/abs/2603.17240",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2603.17240",
    "pdfUrl": "https://arxiv.org/pdf/2603.17240",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{gigaworld_policy,\n      title={GigaWorld-Policy: An Efficient Action-Centered World--Action Model}, \n      author={Angen Ye and Boyuan Wang and Chaojun Ni and Guan Huang and Guosheng Zhao and Hao Li and Hengtao Li and Jie Li and Jindi Lv and Jingyu Liu and Min Cao and Peng Li and Qiuping Deng and Wenjun Mei and Xiaofeng Wang and Xinze Chen and Xinyu Zhou and Yang Wang and Yifan Chang and Yifan Li and Yukun Zhou and Yun Ye and Zhichao Liu and Zheng Zhu},\n      year={2026},\n      journal={arXiv:2603.17240},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "联合视频动作建模",
      "高效推理与实时控制"
    ],
    "architecture": "One Model",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2603.16860",
    "title": "DreamPlan: Efficient Reinforcement Fine-Tuning of Vision-Language Planners via Video World Models",
    "authors": "Emily Yue-Ting Jia; Weiduo Yuan; Tianheng Shi; Vitor Guizilini; Jiageng Mao; Yue Wang",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2026-03-17",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "dreamplan",
    "arxivUrl": "https://arxiv.org/abs/2603.16860",
    "codeUrls": [],
    "projectUrl": "https://psi-lab.ai/DreamPlan/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2603.16860",
    "pdfUrl": "https://arxiv.org/pdf/2603.16860",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{dreamplan,\n      title={DreamPlan: Efficient Reinforcement Fine-Tuning of Vision-Language Planners via Video World Models}, \n      author={Emily Yue-Ting Jia and Weiduo Yuan and Tianheng Shi and Vitor Guizilini and Jiageng Mao and Yue Wang},\n      year={2026},\n      journal={arXiv:2603.16860},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "策略后训练与WM-RL"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2603.16666",
    "title": "Fast-WAM: Do World Action Models Need Test-time Future Imagination?",
    "authors": "Tianyuan Yuan; Zibin Dong; Yicheng Liu; Hang Zhao",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2026-03-17",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "fastwam",
    "arxivUrl": "https://arxiv.org/abs/2603.16666",
    "codeUrls": [
      "https://github.com/yuantianyuan01/FastWAM"
    ],
    "projectUrl": "https://yuantianyuan01.github.io/FastWAM/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2603.16666",
    "pdfUrl": "https://arxiv.org/pdf/2603.16666",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{fastwam,\n      title={Fast-WAM: Do World Action Models Need Test-time Future Imagination?}, \n      author={Tianyuan Yuan and Zibin Dong and Yicheng Liu and Hang Zhao},\n      year={2026},\n      journal={arXiv:2603.16666},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "高效推理与实时控制",
      "潜空间预测与JEPA"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2603.16195",
    "title": "S-VAM: Shortcut Video-Action Model by Self-Distilling Geometric and Semantic Foresight",
    "authors": "Haodong Yan; Zhide Zhong; Jiaguan Zhu; Junjie He; Weilin Yuan; Wenxuan Song; Xin Gong; Yingjie Cai; Guanyi Zhao; Xu Yan; Bingbing Liu; Ying-Cong Chen; Haoang Li",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2026-03-17",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "svam",
    "arxivUrl": "https://arxiv.org/abs/2603.16195",
    "codeUrls": [
      "https://github.com/Haodong-Yan/S-VAM"
    ],
    "projectUrl": "https://haodong-yan.github.io/S-VAM/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2603.16195",
    "pdfUrl": "https://arxiv.org/pdf/2603.16195",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{svam,\n      title={S-VAM: Shortcut Video-Action Model by Self-Distilling Geometric and Semantic Foresight}, \n      author={Haodong Yan and Zhide Zhong and Jiaguan Zhu and Junjie He and Weilin Yuan and Wenxuan Song and Xin Gong and Yingjie Cai and Guanyi Zhao and Xu Yan and Bingbing Liu and Ying-Cong Chen and Haoang Li},\n      year={2026},\n      journal={arXiv:2603.16195},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "视觉规划与IDM",
      "高效推理与实时控制",
      "三维多视角建模"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2603.14948",
    "title": "Bridging Scene Generation and Planning: Driving with World Model via Unifying Vision and Motion Representation",
    "authors": "Xingtai Gui; Meijie Zhang; Tianyi Yan; Wencheng Han; Jiahao Gong; Feiyang Tan; Cheng-zhong Xu; Jianbing Shen",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2026-03-16",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "worlddrive",
    "arxivUrl": "https://arxiv.org/abs/2603.14948",
    "codeUrls": [
      "https://github.com/TabGuigui/WorldDrive"
    ],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2603.14948",
    "pdfUrl": "https://arxiv.org/pdf/2603.14948",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{worlddrive,\n      title={Bridging Scene Generation and Planning: Driving with World Model via Unifying Vision and Motion Representation}, \n      author={Xingtai Gui and Meijie Zhang and Tianyi Yan and Wencheng Han and Jiahao Gong and Feiyang Tan and Cheng-zhong Xu and Jianbing Shen},\n      year={2026},\n      journal={arXiv:2603.14948},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "自动驾驶",
      "潜空间预测与JEPA"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2603.14482",
    "title": "V-JEPA 2.1: Unlocking Dense Features in Video Self-Supervised Learning",
    "authors": "Lorenzo Mur-Labadia; Matthew Muckley; Amir Bar; Mido Assran; Koustuv Sinha; Mike Rabbat; Yann LeCun; Nicolas Ballas; Adrien Bardes",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2026-03-15",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "vjepa2_1",
    "arxivUrl": "https://arxiv.org/abs/2603.14482",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2603.14482",
    "pdfUrl": "https://arxiv.org/pdf/2603.14482",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{vjepa2_1,\n  title={V-JEPA 2.1: Unlocking Dense Features in Video Self-Supervised Learning},\n  author={Mur-Labadia, Lorenzo and Muckley, Matthew and Bar, Amir and Assran, Mahmoud and\nSinha, Koustuv and Rabbat, Michael and LeCun, Yann and Ballas, Nicolas and Bardes, Adrien},\n  journal={arXiv preprint arXiv:2603.14482},\n  year={2026}\n}",
    "majorCategory": "WAM Components",
    "subcategories": [
      "视觉编码器与表征"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2603.19312",
    "title": "LeWorldModel: Stable End-to-End Joint-Embedding Predictive Architecture from Pixels",
    "authors": "Lucas Maes; Quentin Le Lidec; Damien Scieur; Yann LeCun; Randall Balestriero",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2026-03-13",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "lewm",
    "arxivUrl": "https://arxiv.org/abs/2603.19312",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2603.19312",
    "pdfUrl": "https://arxiv.org/pdf/2603.19312",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{lewm,\n      title={LeWorldModel: Stable End-to-End Joint-Embedding Predictive Architecture from Pixels}, \n      author={Lucas Maes and Quentin Le Lidec and Damien Scieur and Yann LeCun and Randall Balestriero},\n      year={2026},\n      journal={arXiv:2603.19312},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "潜空间预测与JEPA",
      "经典WM与模型式RL"
    ],
    "architecture": "不适用",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2603.10448",
    "title": "DiT4DiT: Jointly Modeling Video Dynamics and Actions for Generalizable Robot Control",
    "authors": "Teli Ma; Jia Zheng; Zifan Wang; Chunli Jiang; Andy Cui; Junwei Liang; Shuo Yang",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2026-03-11",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "ma2026dit4dit",
    "arxivUrl": "https://arxiv.org/abs/2603.10448",
    "codeUrls": [
      "https://github.com/Mondo-Robotics/DiT4DiT"
    ],
    "projectUrl": "https://dit4dit.github.io/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2603.10448",
    "pdfUrl": "https://arxiv.org/pdf/2603.10448",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{ma2026dit4dit,\n      title={Dit4dit: Jointly modeling video dynamics and actions for generalizable robot control}, \n      author={Ma, Teli and Zheng, Jia and Wang, Zifan and Jiang, Chunli and Cui, Andy and Liang, Junwei and Yang, Shuo},\n      year={2026},\n      journal={arXiv:2603.10448},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "视觉规划与IDM",
      "泛化与动作对齐"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2603.10422",
    "title": "World2Act: Latent Action Post-Training from World Model Dynamics",
    "authors": "An Dinh Vuong; Tuan Van Vo; Abdullah Sohail; Haoran Ding; Liang Ma; Xiaodan Liang; Anqing Duan; Ivan Laptev; Ian Reid",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2026-03-11",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "world2act",
    "arxivUrl": "https://arxiv.org/abs/2603.10422",
    "codeUrls": [],
    "projectUrl": "https://wm2act.github.io/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2603.10422",
    "pdfUrl": "https://arxiv.org/pdf/2603.10422",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{world2act,\n      title={World2Act: Latent Action Post-Training from World Model Dynamics}, \n      author={An Dinh Vuong and Tuan Van Vo and Abdullah Sohail and Haoran Ding and Liang Ma and Xiaodan Liang and Anqing Duan and Ivan Laptev and Ian Reid},\n      year={2026},\n      journal={arXiv:2603.10422},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "策略后训练与WM-RL",
      "泛化与动作对齐"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2603.08113",
    "title": "SAMoE-VLA: A Scene Adaptive Mixture-of-Experts Vision-Language-Action Model for Autonomous Driving",
    "authors": "Zihan You; Hongwei Liu; Chenxu Dang; Zhe Wang; Sining Ang; Aoqi Wang; Yan Wang",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2026-03-09",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "samoe_vla",
    "arxivUrl": "https://arxiv.org/abs/2603.08113",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2603.08113",
    "pdfUrl": "https://arxiv.org/pdf/2603.08113",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{samoe_vla,\n      title={SAMoE-VLA: A Scene Adaptive Mixture-of-Experts Vision-Language-Action Model for Autonomous Driving}, \n      author={You, Zihan and Liu, Hongwei and Dang, Chenxu and Wang, Zhe and Ang, Sining and Wang, Aoqi and Wang, Yan},\n      year={2026},\n      journal={arXiv:2603.08113},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "自动驾驶",
      "潜空间预测与JEPA",
      "视觉规划与IDM"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2603.03596",
    "title": "MEM: Multi-Scale Embodied Memory for Vision Language Action Models",
    "authors": "Marcel Torne; Karl Pertsch; Homer Walke; Kyle Vedder; Suraj Nair; Brian Ichter; Allen Z. Ren; Haohuan Wang; Jiaming Tang; Kyle Stachowicz; Karan Dhabalia; Michael Equi; Quan Vuong; Jost Tobias Springenberg; Sergey Levine; Chelsea Finn; Danny Driess",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2026-03-04",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "torne2026mem",
    "arxivUrl": "https://arxiv.org/abs/2603.03596",
    "codeUrls": [],
    "projectUrl": "https://www.pi.website/research/memory",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2603.03596",
    "pdfUrl": "https://arxiv.org/pdf/2603.03596",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{torne2026mem,\n  title={MEM: Multi-Scale Embodied Memory for Vision Language Action Models},\n  author={Torne, Marcel and Pertsch, Karl and Walke, Homer and Vedder, Kyle and Nair, Suraj and Ichter, Brian and Ren, Allen Z. and Wang, Haohuan and Tang, Jiaming and Stachowicz, Kyle and Dhabalia, Karan and Equi, Michael and Vuong, Quan and Springenberg, Jost Tobias and Levine, Sergey and Finn, Chelsea and Driess, Danny},\n  journal={arXiv:2603.03596},\n  year={2026}\n}",
    "majorCategory": "VLA",
    "subcategories": [
      "记忆与长时序VLA",
      "分层与双系统VLA"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2603.01441",
    "title": "Unifying Language-Action Understanding and Generation for Autonomous Driving",
    "authors": "Xinyang Wang; Qian Liu; Wenjie Ding; Zhao Yang; Wei Li; Chang Liu; Bailin Li; Kun Zhan; Xianpeng Lang; Wei Chen",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2026-03-02",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "language_action_ad",
    "arxivUrl": "https://arxiv.org/abs/2603.01441",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2603.01441",
    "pdfUrl": "https://arxiv.org/pdf/2603.01441",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{language_action_ad,\n      title={Unifying Language-Action Understanding and Generation for Autonomous Driving}, \n      author={Xinyang Wang and Qian Liu and Wenjie Ding and Zhao Yang and Wei Li and Chang Liu and Bailin Li and Kun Zhan and Xianpeng Lang and Wei Chen},\n      year={2026},\n      journal={arXiv:2603.01441},\n}",
    "majorCategory": "VLA",
    "subcategories": [
      "自回归VLA",
      "自动驾驶VLA",
      "高效推理VLA"
    ],
    "architecture": "One Model",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2602.21633",
    "title": "Self-Correcting VLA: Online Action Refinement via Sparse World Imagination",
    "authors": "Chenyv Liu; Wentao Tan; Lei Zhu; Fengling Li; Jingjing Li; Guoli Yang; Heng Tao Shen",
    "affiliations": "Tongji University; University of Technology Sydney; University of Electronic Science and Technology of China; Advanced Institute of Big Data",
    "contribution": "SC-VLA adds progress and end-effector-change predictions to a GR00T N1.5-based flow policy, then freezes it and trains a SAC residual controller. Predicted motion supplies a directional reward whose influence decreases with predicted progress. Simulation supports both stages; physical ARX5 trials test only the predictive base policy. The evidence supports improved executed manipulation under the reported protocols, with unresolved reward and evaluation details.",
    "abstract": "",
    "submittedDate": "2026-02-25",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv260221633",
    "arxivUrl": "https://arxiv.org/abs/2602.21633",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2602.21633",
    "pdfUrl": "https://arxiv.org/pdf/2602.21633",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "潜空间预测与JEPA",
      "高效推理与实时控制"
    ],
    "architecture": null,
    "predictionParadigm": null,
    "quadrant": null,
    "classificationStatus": null
  },
  {
    "id": "2602.16710",
    "title": "EgoScale: Scaling Dexterous Manipulation with Diverse Egocentric Human Data",
    "authors": "Ruijie Zheng; Dantong Niu; Yuqi Xie; Jing Wang; Mengda Xu; Yunfan Jiang; Fernando Castañeda; Fengyuan Hu; You Liang Tan; Letian Fu; Trevor Darrell; Furong Huang; Yuke Zhu; Danfei Xu; Linxi Fan",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2026-02-18",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "egoscale",
    "arxivUrl": "https://arxiv.org/abs/2602.16710",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2602.16710",
    "pdfUrl": "https://arxiv.org/pdf/2602.16710",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{egoscale,\n      title={EgoScale: Scaling Dexterous Manipulation with Diverse Egocentric Human Data}, \n      author={Ruijie Zheng and Dantong Niu and Yuqi Xie and Jing Wang and Mengda Xu and Yunfan Jiang and Fernando Castañeda and Fengyuan Hu and You Liang Tan and Letian Fu and Trevor Darrell and Furong Huang and Yuke Zhu and Danfei Xu and Linxi Fan},\n      year={2026},\n      journal={arXiv:2602.16710},\n}",
    "majorCategory": "VLA",
    "subcategories": [
      "VLA后训练与数据增强",
      "潜动作预训练",
      "分层与双系统VLA"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2602.16675",
    "title": "Learning to unfold cloth: Scaling up world models to deformable object manipulation",
    "authors": "Jack Rome; Stephen James; Subramanian Ramamoorthy",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2026-02-18",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "cloth_wm",
    "arxivUrl": "https://arxiv.org/abs/2602.16675",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2602.16675",
    "pdfUrl": "https://arxiv.org/pdf/2602.16675",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{cloth_wm,\n      title={Learning to unfold cloth: Scaling up world models to deformable object manipulation}, \n      author={Jack Rome and Stephen James and Subramanian Ramamoorthy},\n      year={2026},\n      journal={arXiv:2602.16675},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "策略后训练与WM-RL",
      "经典WM与模型式RL"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2602.16229",
    "title": "Factored Latent Action World Models",
    "authors": "Zizhao Wang; Chang Shi; Jiaheng Hu; Kevin Rohling; Roberto Martín-Martín; Amy Zhang; Peter Stone",
    "affiliations": "University of Texas at Austin; Sony AI",
    "contribution": "FLAM learns a video world model whose slots each carry a latent action while sharing interaction-aware dynamics networks. Prediction-trained factorization improves rollouts supplied with future-inferred actions and can supply pseudo action labels for behavior cloning. Its strongest evidence concerns multi-entity video modeling; downstream control is evaluated separately in Procgen.",
    "abstract": "",
    "submittedDate": "2026-02-18",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv260216229",
    "arxivUrl": "https://arxiv.org/abs/2602.16229",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2602.16229",
    "pdfUrl": "https://arxiv.org/pdf/2602.16229",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "",
    "majorCategory": "Related resources",
    "subcategories": [
      "Latent-action & representation methods"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": null
  },
  {
    "id": "2602.13977",
    "title": "WoVR: World Models as Reliable Simulators for Post-Training VLA Policies with RL",
    "authors": "Zhennan Jiang; Shangqing Zhou; Yutong Jiang; Zefang Huang; Mingjie Wei; Yuhui Chen; Tianxing Zhou; Zhen Guo; Hao Lin; Quanlu Zhang; Yu Wang; Haoran Li; Chao Yu; Dongbin Zhao",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2026-02-15",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "wovr",
    "arxivUrl": "https://arxiv.org/abs/2602.13977",
    "codeUrls": [],
    "projectUrl": "https://wovr-corl.github.io",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2602.13977",
    "pdfUrl": "https://arxiv.org/pdf/2602.13977",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{wovr,\n      title={WoVR: World Models as Reliable Simulators for Post-Training VLA Policies with RL}, \n      author={Zhennan Jiang and Shangqing Zhou and Yutong Jiang and Zefang Huang and Mingjie Wei and Yuhui Chen and Tianxing Zhou and Zhen Guo and Hao Lin and Quanlu Zhang and Yu Wang and Haoran Li and Chao Yu and Dongbin Zhao},\n      year={2026},\n      journal={arXiv:2602.13977},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "策略后训练与WM-RL",
      "记忆与长时序"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2602.12215",
    "title": "LDA-1B: Scaling Latent Dynamics Action Model via Universal Embodied Data Ingestion",
    "authors": "Jiangran Lyu; Kai Liu; Xuheng Zhang; Haoran Liao; Yusen Feng; Wenxuan Zhu; Tingrui Shen; Jiayi Chen; Jiazhao Zhang; Yifei Dong; Wenbo Cui; Senmao Qi; Shuo Wang; Yixin Zheng; Mi Yan; Xuesong Shi; Haoran Li; Dongbin Zhao; Ming-Yu Liu; Zhizheng Zhang; Li Yi; Yizhou Wang; He Wang",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2026-02-12",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "lda1b",
    "arxivUrl": "https://arxiv.org/abs/2602.12215",
    "codeUrls": [
      "https://github.com/jiangranlv/LDA-1B"
    ],
    "projectUrl": "https://pku-epic.github.io/LDA/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2602.12215",
    "pdfUrl": "https://arxiv.org/pdf/2602.12215",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{lda1b,\n      title={LDA-1B: Scaling Latent Dynamics Action Model via Universal Embodied Data Ingestion}, \n      author={Jiangran Lyu and Kai Liu and Xuheng Zhang and Haoran Liao and Yusen Feng and Wenxuan Zhu and Tingrui Shen and Jiayi Chen and Jiazhao Zhang and Yifei Dong and Wenbo Cui and Senmao Qi and Shuo Wang and Yixin Zheng and Mi Yan and Xuesong Shi and Haoran Li and Dongbin Zhao and Ming-Yu Liu and Zhizheng Zhang and Li Yi and Yizhou Wang and He Wang},\n      year={2026},\n      journal={arXiv:2602.12215},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "潜空间预测与JEPA",
      "联合视频动作建模"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "待核实",
    "quadrant": "待核实",
    "classificationStatus": "部分待核实"
  },
  {
    "id": "2602.12099",
    "title": "GigaBrain-0.5M*: a VLA That Learns From World Model-Based Reinforcement Learning",
    "authors": "GigaBrain Team; Boyuan Wang; Bohan Li; Chaojun Ni; Guan Huang; Guosheng Zhao; Hao Li; Jie Li; Jindi Lv; Jingyu Liu; Lv Feng; Mingming Yu; Peng Li; Qiuping Deng; Tianze Liu; Xinyu Zhou; Xinze Chen; Xiaofeng Wang; Yang Wang; Yifan Li; Yifei Nie; Yilong Li; Yukun Zhou; Yun Ye; Zhichao Liu; Zheng Zhu",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2026-02-12",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "gigabrain0_5",
    "arxivUrl": "https://arxiv.org/abs/2602.12099",
    "codeUrls": [
      "https://github.com/open-gigaai/giga-brain-0"
    ],
    "projectUrl": "https://gigabrain05m.github.io/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2602.12099",
    "pdfUrl": "https://arxiv.org/pdf/2602.12099",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{gigabrain0_5,\n  title={Gigabrain-0.5 m*: a vla that learns from world model-based reinforcement learning},\n  author={Team, GigaBrain and Wang, Boyuan and Li, Bohan and Ni, Chaojun and Huang, Guan and Zhao, Guosheng and Li, Hao and Li, Jie and Lv, Jindi and Liu, Jingyu and others},\n  journal={arXiv:2602.12099},\n  year={2026}\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "策略后训练与WM-RL"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2602.11075",
    "title": "RISE: Self-Improving Robot Policy with Compositional World Model",
    "authors": "Jiazhi Yang; Kunyang Lin; Jinwei Li; Wencong Zhang; Tianwei Lin; Longyan Wu; Zhizhong Su; Hao Zhao; Ya-Qin Zhang; Li Chen; Ping Luo; Xiangyu Yue; Hongyang Li",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2026-02-11",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "rise",
    "arxivUrl": "https://arxiv.org/abs/2602.11075",
    "codeUrls": [
      "https://github.com/OpenDriveLab/RISE"
    ],
    "projectUrl": "https://opendrivelab.com/RISE/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2602.11075",
    "pdfUrl": "https://arxiv.org/pdf/2602.11075",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{rise,\n      title={RISE: Self-Improving Robot Policy with Compositional World Model}, \n      author={Jiazhi Yang and Kunyang Lin and Jinwei Li and Wencong Zhang and Tianwei Lin and Longyan Wu and Zhizhong Su and Hao Zhao and Ya-Qin Zhang and Li Chen and Ping Luo and Xiangyu Yue and Hongyang Li},\n      year={2026},\n      journal={arXiv:2602.11075},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "策略后训练与WM-RL",
      "三维多视角建模"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2602.10102",
    "title": "VideoWorld 2: Learning Transferable Knowledge from Real-world Videos",
    "authors": "Zhongwei Ren; Yunchao Wei; Xiao Yu; Guixun Luo; Yao Zhao; Bingyi Kang; Jiashi Feng; Xiaojie Jin",
    "affiliations": "ByteDance Seed; Beijing Jiaotong University",
    "contribution": "VideoWorld 2 learns discrete visual-dynamics codes with a pretrained diffusion appearance prior, then trains an autoregressive transformer to predict those codes. It improves generated long-horizon craft sequences and transfers latent pretraining to a separately action-supervised CALVIN policy. Its evidence supports benchmark-specific transfer; generated craft success, simulator control and complete appearance disentanglement remain distinct claims.",
    "abstract": "",
    "submittedDate": "2026-02-10",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv260210102",
    "arxivUrl": "https://arxiv.org/abs/2602.10102",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2602.10102",
    "pdfUrl": "https://arxiv.org/pdf/2602.10102",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "",
    "majorCategory": "VLA",
    "subcategories": [
      "潜动作预训练"
    ],
    "architecture": null,
    "predictionParadigm": null,
    "quadrant": null,
    "classificationStatus": null
  },
  {
    "id": "2602.10098",
    "title": "VLA-JEPA: Enhancing Vision-Language-Action Model with Latent World Model",
    "authors": "Jingwen Sun; Wenyao Zhang; Zekun Qi; Shaojie Ren; Zezhi Liu; Hanxin Zhu; Guangzhong Sun; Xin Jin; Zhibo Chen",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2026-02-10",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "vla_jepa",
    "arxivUrl": "https://arxiv.org/abs/2602.10098",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2602.10098",
    "pdfUrl": "https://arxiv.org/pdf/2602.10098",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{vla_jepa,\n      title={VLA-JEPA: Enhancing Vision-Language-Action Model with Latent World Model}, \n      author={Jingwen Sun and Wenyao Zhang and Zekun Qi and Shaojie Ren and Zezhi Liu and Hanxin Zhu and Guangzhong Sun and Xin Jin and Zhibo Chen},\n      year={2026},\n      journal={arXiv:2602.10098},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "潜空间预测与JEPA",
      "泛化与动作对齐"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "联合预测",
    "quadrant": "Q3 · Dual-system × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2602.08971",
    "title": "WorldArena: A Unified Benchmark for Evaluating Perception and Functional Utility of Embodied World Models",
    "authors": "Yu Shang; Zhuohang Li; Yiding Ma; Weikang Su; Xin Jin; Ziyou Wang; Lei Jin; Xin Zhang; Yinzhou Tang; Haisheng Su; Chen Gao; Wei Wu; Xihui Liu; Dhruv Shah; Zhaoxiang Zhang; Zhibo Chen; Jun Zhu; Yonghong Tian; Tat-Seng Chua; Wenwu Zhu; Yong Li",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2026-02-09",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "worldarena",
    "arxivUrl": "https://arxiv.org/abs/2602.08971",
    "codeUrls": [
      "https://github.com/tsinghua-fib-lab/WorldArena"
    ],
    "projectUrl": "https://world-arena.ai",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2602.08971",
    "pdfUrl": "https://arxiv.org/pdf/2602.08971",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{worldarena,\n      title={WorldArena: A Unified Benchmark for Evaluating Perception and Functional Utility of Embodied World Models}, \n      author={Yu Shang and Zhuohang Li and Yiding Ma and Weikang Su and Xin Jin and Ziyou Wang and Lei Jin and Xin Zhang and Yinzhou Tang and Haisheng Su and Chen Gao and Wei Wu and Xihui Liu and Dhruv Shah and Zhaoxiang Zhang and Zhibo Chen and Jun Zhu and Yonghong Tian and Tat-Seng Chua and Wenwu Zhu and Yong Li},\n      year={2026},\n      journal={arXiv:2602.08971},\n}",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "世界模型评测基准",
      "评测协议与诊断",
      "综合评分"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2602.06508",
    "title": "World-VLA-Loop: Closed-Loop Learning of Video World Model and VLA Policy",
    "authors": "Xiaokang Liu; Zechen Bai; Hai Ci; Kevin Yuchen Ma; Mike Zheng Shou",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2026-02-06",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "world_vla_loop",
    "arxivUrl": "https://arxiv.org/abs/2602.06508",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2602.06508",
    "pdfUrl": "https://arxiv.org/pdf/2602.06508",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{world_vla_loop,\n      title={World-VLA-Loop: Closed-Loop Learning of Video World Model and VLA Policy}, \n      author={Xiaokang Liu and Zechen Bai and Hai Ci and Kevin Yuchen Ma and Mike Zheng Shou},\n      year={2026},\n      journal={arXiv:2602.06508},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "策略后训练与WM-RL"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2602.06001",
    "title": "Visuo-Tactile World Models",
    "authors": "Carolina Higuera; Sergio Arnaud; Byron Boots; Mustafa Mukadam; Francois Robert Hogan; Franziska Meier",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2026-02-05",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "vt_wm",
    "arxivUrl": "https://arxiv.org/abs/2602.06001",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2602.06001",
    "pdfUrl": "https://arxiv.org/pdf/2602.06001",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{vt_wm,\n      title={Visuo-Tactile World Models}, \n      author={Carolina Higuera and Sergio Arnaud and Byron Boots and Mustafa Mukadam and Francois Robert Hogan and Franziska Meier},\n      year={2026},\n      journal={arXiv:2602.06001},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "多模态触觉音频",
      "潜空间预测与JEPA"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2602.02473",
    "title": "HumanX: Toward Agile and Generalizable Humanoid Interaction Skills from Human Videos",
    "authors": "Yinhuai Wang; Qihan Zhao; Yuen Fui Lau; Runyi Yu; Hok Wai Tsui; Qifeng Chen; Jingbo Wang; Jiangmiao Pang; Ping Tan",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2026-02-02",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "Wang2026HumanXAgile",
    "arxivUrl": "https://arxiv.org/abs/2602.02473",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2602.02473",
    "pdfUrl": "https://arxiv.org/pdf/2602.02473",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{Wang2026HumanXAgile,\n      title={HumanX: Toward Agile and Generalizable Humanoid Interaction Skills from Human Videos}, \n      author={Yinhuai Wang and Qihan Zhao and Yuen Fui Lau and Runyi Yu and Hok Wai Tsui and Qifeng Chen and Jingbo Wang and Jiangmiao Pang and Ping Tan},\n      year={2026},\n      journal={arXiv:2602.02473},\n}",
    "majorCategory": "数据集",
    "subcategories": [
      "合成数据与数据生成",
      "示范数据"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2601.21998",
    "title": "Causal World Modeling for Robot Control",
    "authors": "Lin Li; Qihang Zhang; Yiming Luo; Shuai Yang; Ruilin Wang; Fei Han; Mingrui Yu; Zelin Gao; Nan Xue; Xing Zhu; Yujun Shen; Yinghao Xu",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2026-01-29",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "lingbotva",
    "arxivUrl": "https://arxiv.org/abs/2601.21998",
    "codeUrls": [
      "https://github.com/robbyant/lingbot-va"
    ],
    "projectUrl": "https://technology.robbyant.com/lingbot-va",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2601.21998",
    "pdfUrl": "https://arxiv.org/pdf/2601.21998",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{lingbotva,\n      title={Causal World Modeling for Robot Control}, \n      author={Lin Li and Qihang Zhang and Yiming Luo and Shuai Yang and Ruilin Wang and Fei Han and Mingrui Yu and Zelin Gao and Nan Xue and Xing Zhu and Yujun Shen and Yinghao Xu},\n      year={2026},\n      journal={arXiv:2601.21998},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "视觉规划与IDM",
      "高效推理与实时控制",
      "记忆与长时序"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2601.16163",
    "title": "Cosmos Policy: Fine-Tuning Video Models for Visuomotor Control and Planning",
    "authors": "Moo Jin Kim; Yihuai Gao; Tsung-Yi Lin; Yen-Chen Lin; Yunhao Ge; Grace Lam; Percy Liang; Shuran Song; Ming-Yu Liu; Chelsea Finn; Jinwei Gu",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2026-01-22",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "cosmos_policy",
    "arxivUrl": "https://arxiv.org/abs/2601.16163",
    "codeUrls": [
      "https://github.com/nvlabs/cosmos-policy"
    ],
    "projectUrl": "https://research.nvidia.com/labs/cosmos-lab/cosmos-policy/",
    "venue": "ICLR 2026",
    "paperUrl": "https://arxiv.org/abs/2601.16163",
    "pdfUrl": "https://arxiv.org/pdf/2601.16163",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{cosmos_policy,\n      title={Cosmos Policy: Fine-Tuning Video Models for Visuomotor Control and Planning}, \n      author={Moo Jin Kim and Yihuai Gao and Tsung-Yi Lin and Yen-Chen Lin and Yunhao Ge and Grace Lam and Percy Liang and Shuran Song and Ming-Yu Liu and Chelsea Finn and Jinwei Gu},\n      year={2026},\n      journal={arXiv:2601.16163},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "联合视频动作建模",
      "泛化与动作对齐"
    ],
    "architecture": "One Model",
    "predictionParadigm": "联合预测",
    "quadrant": "Q1 · One Model × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2601.04453",
    "title": "UniDrive-WM: Unified Understanding, Planning and Generation World Model for Autonomous Driving",
    "authors": "Zhexiao Xiong; Xin Ye; Burhan Yaman; Sheng Cheng; Yiren Lu; Jingru Luo; Nathan Jacobs; Liu Ren",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2026-01-07",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "unidrive_wm",
    "arxivUrl": "https://arxiv.org/abs/2601.04453",
    "codeUrls": [],
    "projectUrl": "https://unidrive-wm.github.io/UniDrive-WM/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2601.04453",
    "pdfUrl": "https://arxiv.org/pdf/2601.04453",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{unidrive_wm,\n      title={UniDrive-WM: Unified Understanding, Planning and Generation World Model For Autonomous Driving}, \n      author={Zhexiao Xiong and Xin Ye and Burhan Yaman and Sheng Cheng and Yiren Lu and Jingru Luo and Nathan Jacobs and Liu Ren},\n      year={2026},\n      journal={arXiv:2601.04453},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "联合视频动作建模",
      "自动驾驶"
    ],
    "architecture": "待核实",
    "predictionParadigm": "联合预测",
    "quadrant": "待核实",
    "classificationStatus": "部分待核实"
  },
  {
    "id": "2601.04137",
    "title": "Wow, wo, val! A Comprehensive Embodied World Model Evaluation Turing Test",
    "authors": "Chun-Kai Fan; Xiaowei Chi; Xiaozhu Ju; Hao Li; Yong Bao; Yu-Kai Wang; Lizhang Chen; Zhiyuan Jiang; Kuangzhi Ge; Ying Li; Weishi Mi; Qingpo Wuwu; Peidong Jia; Yulin Luo; Kevin Zhang; Zhiyuan Qin; Yong Dai; Sirui Han; Yike Guo; Shanghang Zhang; Jian Tang",
    "affiliations": "State Key Laboratory of Multimedia Information Processing, School of Computer Science, Peking University; Beijing Innovation Center of Humanoid Robotics; The Hong Kong University of Science and Technology",
    "contribution": "WoW-World-Eval tests whether instruction-conditioned robot videos are visually convincing, task-correct, physically plausible and usable for action extraction. Its 609-sample benchmark combines automated metrics, human judgments and a separate GC-IDM robot-execution test. Hailuo leads the reported aggregate video score, while WoW-wan leads physical execution. The central lesson is that a plausible imagined manipulation and an executable one are distinct outcomes; calibration and evaluator dependence constrain how broadly the scores can be interpreted.",
    "abstract": "",
    "submittedDate": "2026-01-07",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv260104137",
    "arxivUrl": "https://arxiv.org/abs/2601.04137",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2601.04137",
    "pdfUrl": "https://arxiv.org/pdf/2601.04137",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "世界模型评测基准",
      "评测协议与诊断"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": null
  },
  {
    "id": "2601.04061",
    "title": "CLAP: Contrastive Latent Action Pretraining for Learning Vision-Language-Action Models from Human Videos",
    "authors": "Chubin Zhang; Jianan Wang; Zifeng Gao; Yue Su; Tianru Dai; Cai Zhou; Jiwen Lu; Yansong Tang",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2026-01-07",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "Zhang2026CLAPContrastive",
    "arxivUrl": "https://arxiv.org/abs/2601.04061",
    "codeUrls": [
      "https://github.com/LinShan-Bin/OpenCLAP"
    ],
    "projectUrl": "https://lin-shan.com/CLAP/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2601.04061",
    "pdfUrl": "https://arxiv.org/pdf/2601.04061",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{Zhang2026CLAPContrastive,\n      title={CLAP: Contrastive Latent Action Pretraining for Learning Vision-Language-Action Models from Human Videos}, \n      author={Chubin Zhang and Jianan Wang and Zifeng Gao and Yue Su and Tianru Dai and Cai Zhou and Jiwen Lu and Yansong Tang},\n      year={2026},\n      journal={arXiv:2601.04061},\n}",
    "majorCategory": "VLA",
    "subcategories": [
      "潜动作预训练",
      "自回归VLA",
      "扩散与流匹配VLA"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2601.03782",
    "title": "PointWorld: Scaling 3D World Models for In-The-Wild Robotic Manipulation",
    "authors": "Wenlong Huang; Yu-Wei Chao; Arsalan Mousavian; Ming-Yu Liu; Dieter Fox; Kaichun Mo; Li Fei-Fei",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2026-01-07",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "pointworld",
    "arxivUrl": "https://arxiv.org/abs/2601.03782",
    "codeUrls": [
      "https://github.com/NVlabs/PointWorld"
    ],
    "projectUrl": "https://point-world.github.io/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2601.03782",
    "pdfUrl": "https://arxiv.org/pdf/2601.03782",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{pointworld,\n      title={PointWorld: Scaling 3D World Models for In-The-Wild Robotic Manipulation}, \n      author={Wenlong Huang and Yu-Wei Chao and Arsalan Mousavian and Ming-Yu Liu and Dieter Fox and Kaichun Mo and Li Fei-Fei},\n      year={2026},\n      journal={arXiv:2601.03782},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "三维多视角建模",
      "泛化与动作对齐"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-fa771119edf30fd77084",
    "title": "Motus: A Unified Latent Action World Model",
    "authors": "Hongzhe Bi; Hengkai Tan; Shenghao Xie; Zeyuan Wang; Shuhe Huang; Haitian Liu; Ruowen Zhao; Yao Feng; Chendong Xiang; Yinze Rong; Hongyan Zhao; Hanyu Liu; Zhizhong Su; Lei Ma; Hang Su; Jun Zhu",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "motus",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "CVPR 2026",
    "paperUrl": "https://openaccess.thecvf.com/content/CVPR2026/html/Bi_Motus_A_Unified_Latent_Action_World_Model_CVPR_2026_paper.html",
    "pdfUrl": "https://arxiv.org/pdf/2512.13030",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@inproceedings{motus,\n  title={Motus: A unified latent action world model},\n  author={Bi, Hongzhe and Tan, Hengkai and Xie, Shenghao and Wang, Zeyuan and Huang, Shuhe and Liu, Haitian and Zhao, Ruowen and Feng, Yao and Xiang, Chendong and Rong, Yinze and others},\n  booktitle={CVPR},\n  pages={35101--35113},\n  year={2026}\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "联合视频动作建模",
      "泛化与动作对齐"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "联合预测",
    "quadrant": "Q3 · Dual-system × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-f66b58acc9c3ad41921f",
    "title": "Vtla: Vision-tactile-language-action model with preference learning for insertion manipulation",
    "authors": "Chaofan Zhang; Peng Hao; Xiaoge Cao; Xiaoshuai Hao; Shaowei Cui; Shuo Wang",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "vtla",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "Biomimetic Intelligence and Robotics, 2026",
    "paperUrl": "https://doi.org/10.1016/j.birob.2026.100333",
    "pdfUrl": "https://arxiv.org/pdf/2505.09577",
    "doi": "https://doi.org/10.1016/j.birob.2026.100333",
    "publicationYear": 2026,
    "bibtex": "@article{vtla,\n  title={Vtla: Vision-tactile-language-action model with preference learning for insertion manipulation},\n  author={Zhang, Chaofan and Hao, Peng and Cao, Xiaoge and Hao, Xiaoshuai and Cui, Shaowei and Wang, Shuo},\n  journal={Biomim. Intell. Robot.},\n  pages={100333},\n  year={2026},\n}",
    "majorCategory": "VLA",
    "subcategories": [
      "多模态触觉VLA",
      "VLA后训练与数据增强",
      "自回归VLA"
    ],
    "architecture": "One Model",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-ed0e9bb8027f431c1f20",
    "title": "Dyna-2: A 1-million-hour scaling law for world-action models",
    "authors": "Dyna Robotics",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "robotics2026dyna",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": "https://www.dyna.co/dyna-2",
    "venue": null,
    "paperUrl": "https://www.dyna.co/dyna-2",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@techreport{robotics2026dyna,\n  author      = {{Dyna Robotics}},\n  title       = {{Dyna-2}: A 1-million-hour scaling law for world-action models},\n  institution = {Dyna Robotics},\n  year        = {2026},\n  type        = {Official Research Report},\n  url         = {https://www.dyna.co/dyna-2}\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "泛化与动作对齐",
      "高效推理与实时控制"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-ced7109d62bb4514d467",
    "title": "Percept-WAM: Perception-Enhanced World-Awareness-Action Model for Robust End-to-End Autonomous Driving",
    "authors": "Jianhua Han; Meng Tian; Jiangtong Zhu; Fan He; Huixin Zhang; Sitong Guo; Dechang Zhu; Hao Tang; Pei Xu; Yuze Guo; Minzhe Niu; Haojie Zhu; Qichao Dong; Xuechao Yan; Siyuan Dong; Lu Hou; Qingqiu Huang; Xiaosong Jia; Hang Xu",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "percept_wam",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "CVPR 2026",
    "paperUrl": "https://openaccess.thecvf.com/content/CVPR2026/html/Han_Percept-WAM_Perception-Enhanced_World-Awareness-Action_Model_for_Robust_End-to-End_Autonomous_Driving_CVPR_2026_paper.html",
    "pdfUrl": "https://openaccess.thecvf.com/content/CVPR2026/papers/Han_Percept-WAM_Perception-Enhanced_World-Awareness-Action_Model_for_Robust_End-to-End_Autonomous_Driving_CVPR_2026_paper.pdf",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@InProceedings{percept_wam,\n    author    = {Han, Jianhua and Tian, Meng and Zhu, Jiangtong and He, Fan and Zhang, Huixin and Guo, Sitong and Zhu, Dechang and Tang, Hao and Xu, Pei and Guo, Yuze and Niu, Minzhe and Zhu, Haojie and Dong, Qichao and Yan, Xuechao and Dong, Siyuan and Hou, Lu and Huang, Qingqiu and Jia, Xiaosong and Xu, Hang},\n    title     = {Percept-WAM: Perception-Enhanced World-Awareness-Action Model for Robust End-to-End Autonomous Driving},\n    booktitle = {Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)},\n    month     = {June},\n    year      = {2026},\n    pages     = {10642-10655}\n}",
    "majorCategory": "VLA",
    "subcategories": [
      "自动驾驶VLA",
      "空间感知增强VLA"
    ],
    "architecture": "One Model",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-c558ebadd23e762f0de7",
    "title": "Ctrl-World: A Controllable Generative World Model for Robot Manipulation",
    "authors": "Yanjiang Guo; Lucy Xiaoyang Shi; Jianyu Chen; Chelsea Finn",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "ctrl_world",
    "arxivUrl": null,
    "codeUrls": [
      "https://github.com/Robert-gyj/Ctrl-World"
    ],
    "projectUrl": "https://ctrl-world.github.io/",
    "venue": "ICLR 2026",
    "paperUrl": "https://openreview.net/forum?id=748bHL2BAv",
    "pdfUrl": "https://arxiv.org/pdf/2510.10125",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@inproceedings{ctrl_world,\n  title={Ctrl-world: A controllable generative world model for robot manipulation},\n  author={Guo, Yanjiang and Shi, Lucy and Chen, Jianyu and Finn, Chelsea},\n  booktitle={ICLR},\n  pages={6121--6138},\n  year={2026}\n}",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "神经世界模拟器"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-c36572e136b3aca5087e",
    "title": "Towards Generalist Embodied AI: A Survey on World Models for VLA Agents",
    "authors": "Wentao Tan; Lei Zhu; Bowen Wang; Enci Xie; Baixu Ji; Zengrong Lin; Wenjie Yang; Jingjing Li; Heng Tao Shen",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "tan2026towards",
    "arxivUrl": null,
    "codeUrls": [
      "https://github.com/FutureTwT/awesome-world-models-for-vla-agents"
    ],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://www.techrxiv.org/doi/full/10.36227/techrxiv.176948355.54623875/v1",
    "pdfUrl": "https://www.techrxiv.org/doi/pdf/10.36227/techrxiv.176948355.54623875/v1",
    "doi": "https://doi.org/10.36227/techrxiv.176948355.54623875/v1",
    "publicationYear": 2026,
    "bibtex": "@article{tan2026towards,\ntitle={Towards generalist embodied ai: A survey on world models for vla agents},\n  author={Tan, Wentao and Zhu, Lei and Wang, Bowen and Xie, Enci and Ji, Baixu and Lin, Zengrong and Yang, Wenjie and Li, Jingjing and Shen, Heng Tao},\n  year={2026},\n  journal={TechRxiv}\n}",
    "majorCategory": "Related resources",
    "subcategories": [
      "综述与技术资源"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "综述明确"
  },
  {
    "id": "ref-c268d974279d36619757",
    "title": "MonoDream: Monocular Vision-Language Navigation with Panoramic Dreaming",
    "authors": "Shuo Wang; Yongcai Wang; Zhaoxin Fan; Yucheng Wang; Maiyue Chen; Kaihui Wang; Zhizhong Su; Wanting Li; Xudong Cai; Yeying Jin; Deying Li",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "monodream",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": "https://horizonrobotics.github.io/robot_lab/monodream/",
    "venue": "AAAI 2026",
    "paperUrl": "https://ojs.aaai.org/index.php/AAAI/article/view/37974",
    "pdfUrl": "https://ojs.aaai.org/index.php/AAAI/article/download/37974/41936",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@inproceedings{monodream,\n  author       = {Shuo Wang and\n                  Yongcai Wang and\n                  Zhaoxin Fan and\n                  Yucheng Wang and\n                  Maiyue Chen and\n                  Kaihui Wang and\n                  Zhizhong Su and\n                  Wanting Li and\n                  Xudong Cai and\n                  Yeying Jin and\n                  Deying Li},\n  title        = {MonoDream: Monocular Vision-Language Navigation with Panoramic Dreaming},\n  booktitle    = {AAAI},\n  pages        = {10074--10082},\n  year         = {2026},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "导航",
      "潜空间预测与JEPA"
    ],
    "architecture": "One Model",
    "predictionParadigm": "联合预测",
    "quadrant": "Q1 · One Model × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-b5386c6f934f87f4cec4",
    "title": "OpenDriveVLA: Towards End-to-end Autonomous Driving with Large Vision Language Action Model",
    "authors": "Xingcheng Zhou; Xuyuan Han; Feng Yang; Yunpu Ma; Volker Tresp; Alois Knoll",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "opendrivevla",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "AAAI 2026",
    "paperUrl": "https://ojs.aaai.org/index.php/AAAI/article/view/38386",
    "pdfUrl": "https://ojs.aaai.org/index.php/AAAI/article/download/38386/42348",
    "doi": "https://doi.org/10.1609/aaai.v40i16.38386",
    "publicationYear": 2026,
    "bibtex": "@inproceedings{opendrivevla,\n  author       = {Xingcheng Zhou and\n                  Xuyuan Han and\n                  Feng Yang and\n                  Yunpu Ma and\n                  Volker Tresp and\n                  Alois Knoll},\n  title        = {OpenDriveVLA: Towards End-to-end Autonomous Driving with Large Vision\n                  Language Action Model},\n  booktitle    = {AAAI},\n  pages        = {13782--13790},\n  year         = {2026},\n}",
    "majorCategory": "VLA",
    "subcategories": [
      "自回归VLA",
      "自动驾驶VLA"
    ],
    "architecture": "One Model",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-b21a967bfc27f43d29b2",
    "title": "DriveLaW: Unifying Planning and Video Generation in a Latent Driving World",
    "authors": "Tianze Xia; Yongkang Li; Lijun Zhou; Jingfeng Yao; Kaixin Xiong; Haiyang Sun; Bing Wang; Kun Ma; Guang Chen; Hangjun Ye; Wenyu Liu; Xinggang Wang",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "drivelaw",
    "arxivUrl": null,
    "codeUrls": [
      "https://github.com/xiaomi-research/drivelaw"
    ],
    "projectUrl": null,
    "venue": "CVPR 2026",
    "paperUrl": "https://openaccess.thecvf.com/content/CVPR2026/html/Xia_DriveLaW_Unifying_Planning_and_Video_Generation_in_a_Latent_Driving_CVPR_2026_paper.html",
    "pdfUrl": "https://openaccess.thecvf.com/content/CVPR2026/papers/Xia_DriveLaW_Unifying_Planning_and_Video_Generation_in_a_Latent_Driving_CVPR_2026_paper.pdf",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@InProceedings{drivelaw,\n    author    = {Xia, Tianze and Li, Yongkang and Zhou, Lijun and Yao, Jingfeng and Xiong, Kaixin and Sun, Haiyang and Wang, Bing and Ma, Kun and Chen, Guang and Ye, Hangjun and Liu, Wenyu and Wang, Xinggang},\n    title     = {DriveLaW: Unifying Planning and Video Generation in a Latent Driving World},\n    booktitle = {CVPR},\n    year      = {2026},\n    pages     = {39701-39712}\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "自动驾驶",
      "视觉规划与IDM"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-95ab3952f53728ca5f82",
    "title": "RoboTwin 2.0: A Scalable Data Generator and Benchmark with Strong Domain Randomization for Robust Bimanual Robotic Manipulation",
    "authors": "Tianxing Chen; Zanxin Chen; Baijun Chen; Zijian Cai; Yibin Liu; Zixuan Li; Qiwei Liang; Xianliang Lin; Yiheng Ge; Zhenyu Gu; Weiliang Deng; Yubin Guo; Tian Nian; Xuanbing Xie; Qiangyu Chen; Kailun Su; Tianling Xu; Guodong Liu; Mengkang Hu; Huan-ang Gao; Kaixuan Wang; Zhixuan Liang; Yusen Qin; Xiaokang Yang; Ping Luo; Yao Mu",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "robotwin2",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": "https://robotwin-platform.github.io/",
    "venue": "ICML 2026",
    "paperUrl": "https://icml.cc/virtual/2026/poster/62192",
    "pdfUrl": "https://arxiv.org/pdf/2506.18088",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@inproceedings{\nrobotwin2,\ntitle={RoboTwin 2.0: A Scalable Data Generator and Benchmark with Strong Domain Randomization for Robust Bimanual Robotic Manipulation},\nauthor={Tianxing Chen and Zanxin Chen and Baijun Chen and Zijian Cai and Yibin Liu and Zixuan Li and Qiwei Liang and Xianliang Lin and Yiheng Ge and Zhenyu Gu and Weiliang Deng and Yubin Guo and Tian Nian and Xuanbing Xie and Qiangyu Chen and Kailun Su and Tianling Xu and Guodong Liu and Mengkang Hu and Huan-ang Gao and Kaixuan Wang and Zhixuan Liang and Yusen Qin and Xiaokang Yang and Ping Luo and Yao Mu},\nbooktitle={ICML},\nyear={2026},\n}",
    "majorCategory": "数据集",
    "subcategories": [
      "合成数据与数据生成",
      "双臂机器人数据",
      "三维物体资源",
      "机器人操作基准",
      "鲁棒性与泛化评测"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-8ee63fbf3242512e5166",
    "title": "Hierarchical Latent Action Model",
    "authors": "Hanjung Kim; Lerrel Pinto; Seon Joo Kim",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "hilam",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "ICLR 2026 Workshop on World Models",
    "paperUrl": "https://openreview.net/forum?id=IUaiouMYXp",
    "pdfUrl": "https://arxiv.org/pdf/2603.05815",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@inproceedings{\nhilam,\ntitle={Hierarchical Latent Action Model},\nauthor={Hanjung Kim and Lerrel Pinto and Seon Joo Kim},\nbooktitle={ICLR Workshop},\nyear={2026},\n}",
    "majorCategory": "VLA",
    "subcategories": [
      "潜动作预训练",
      "动作策略基础"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-7e3d0c035bab4ff2d644",
    "title": "DreamDojo: A Real-Time Robot World Model from Large-Scale Human Videos",
    "authors": "Shenyuan Gao; William Liang; Kaiyuan Zheng; Ayaan Malik; Seonghyeon Ye; Sihyun Yu; Wei-Cheng Tseng; Yuzhu Dong; Kaichun Mo; Chen-Hsuan Lin; Qianli Ma; Seungjun Nah; Loic Magne; Jiannan Xiang; Yuqi Xie; Ruijie Zheng; Dantong Niu; You Liang Tan; K. R. Zentner; George Kurian; Suneel Indupuru; Pooya Jannaty; Jinwei Gu; Jun Zhang; Jitendra Malik; Pieter Abbeel; Ming-Yu Liu; Yuke Zhu; Joel Jang; Linxi \"Jim\" Fan",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "dreamdojo",
    "arxivUrl": null,
    "codeUrls": [
      "https://github.com/NVIDIA/DreamDojo"
    ],
    "projectUrl": "https://dreamdojo-world.github.io/",
    "venue": "ICML 2026",
    "paperUrl": "https://icml.cc/virtual/2026/poster/65193",
    "pdfUrl": "https://arxiv.org/pdf/2602.06949",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@inproceedings{dreamdojo,\ntitle={DreamDojo: A Generalist Robot World Model from Large-Scale Human Videos},\nauthor={Shenyuan Gao and William Liang and Kaiyuan Zheng and Ayaan Naveed Malik and Seonghyeon Ye and Sihyun Yu and Wei-Cheng Tseng and Yuzhu Dong and Kaichun Mo and Chen-Hsuan Lin and Jiannan Xiang and Yuqi Xie and Ruijie Zheng and Dantong Niu and Pooya Jannaty and Jinwei Gu and Jun Zhang and Jitendra Malik and Pieter Abbeel and Ming-Yu Liu and Yuke Zhu and Joel Jang and Linxi Fan},\nbooktitle={ICML},\nyear={2026},\n\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "神经世界模拟器",
      "潜动作预训练"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-7a6225623e6351324a10",
    "title": "Learning Physics from Pretrained Video Models: A Multimodal Continuous and Sequential World Interaction Models for Robotic Manipulation",
    "authors": "Zijian Song; Qichang Li; Sihan Qin; Yuhao Chen; Tianshui Chen; Liang Lin; Guangrun Wang",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "physgen",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "ICMR 2026",
    "paperUrl": "https://dl.acm.org/doi/10.1145/3805622.3810752",
    "pdfUrl": "https://arxiv.org/pdf/2603.00110",
    "doi": "https://doi.org/10.1145/3805622.3810752",
    "publicationYear": 2026,
    "bibtex": "@inproceedings{physgen,\n  title={Learning physics from pretrained video models: A multimodal continuous and sequential world interaction models for robotic manipulation},\n  author={Song, Zijian and Li, Qichang and Qin, Sihan and Chen, Yuhao and Chen, Tianshui and Lin, Liang and Wang, Guangrun},\n  booktitle={ICMR},\n  pages={758--767},\n  year={2026}\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "视觉规划与IDM",
      "高效推理与实时控制"
    ],
    "architecture": "One Model",
    "predictionParadigm": "IDM",
    "quadrant": "Q2 · One Model × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-75da67884f9b1e1a961e",
    "title": "WorldGym: World Model as An Environment for Policy Evaluation",
    "authors": "Julian Quevedo; Ansh Kumar Sharma; Yixiang Sun; Varad Suryavanshi; Percy Liang; Sherry Yang",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "worldgym",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "ICLR 2026",
    "paperUrl": "https://proceedings.iclr.cc/paper_files/paper/2026/hash/7f5e909ac0324db03506b380c695ffaf-Abstract-Conference.html",
    "pdfUrl": "https://openreview.net/pdf?id=hidBHy1CAw",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@inproceedings{worldgym,\n  title={Worldgym: World model as an environment for policy evaluation},\n  author={Quevedo, Julian and Sharma, Ansh Kumar and Sun, Yixiang and Suryavanshi, Varad and Liang, Percy and Yang, Sherry},\n  booktitle={ICLR},\n  volume={2026},\n  pages={78932--78957},\n  year={2026}\n}",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "神经策略评测环境",
      "机器人操作基准"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "综述明确"
  },
  {
    "id": "ref-6c70da879cfe4f99ac28",
    "title": "AutoMoT: A Unified Vision-Language-Action Model with Asynchronous Mixture-of-Transformers for End-to-End Autonomous Driving",
    "authors": "Wenhui Huang; Songyan Zhang; Qihang Huang; Zhidong Wang; Zhiqi Mao; Collister Chua; Zhan Chen; Long Chen; Chen Lv",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "automot",
    "arxivUrl": null,
    "codeUrls": [
      "https://github.com/OscarHuangWind/AutoMoT"
    ],
    "projectUrl": "https://automot-website.github.io/",
    "venue": "ICML 2026",
    "paperUrl": "https://icml.cc/virtual/2026/poster/64489",
    "pdfUrl": "https://arxiv.org/pdf/2603.14851",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@inproceedings{\nautomot,\ntitle={AutoMoT: A Unified Vision-Language-Action Model with Asynchronous Mixture -of-Transformers for End-to-End Autonomous Driving},\nauthor={Wenhui Huang and Songyan Zhang and Qihang Huang and Zhidong Wang and Zhiqi Mao and Collister Chua and Zhan Chen and Long Chen and Chen Lv},\nbooktitle={ICML},\nyear={2026},\n\n}",
    "majorCategory": "VLA",
    "subcategories": [
      "分层与双系统VLA"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-6b83535e39c8daa32058",
    "title": "VLAW: Iterative Co-Improvement of Vision-Language-Action Policy and World Model",
    "authors": "Yanjiang Guo; Tony Lee; Lucy Xiaoyang Shi; Jianyu Chen; Percy Liang; Chelsea Finn",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "vlaw",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": "https://sites.google.com/view/vlaw-arxiv",
    "venue": "ICML 2026",
    "paperUrl": "https://icml.cc/virtual/2026/poster/66169",
    "pdfUrl": "https://arxiv.org/pdf/2602.12063",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@inproceedings{\nvlaw,\ntitle={{VLAW}: Iterative Co-Improvement of Vision-Language-Action Policy and World Model},\nauthor={Yanjiang Guo and Tony Lee and Lucy Xiaoyang Shi and Jianyu Chen and Percy Liang and Chelsea Finn},\nbooktitle={ICML},\nyear={2026},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "策略后训练与WM-RL"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-619b5a5b1bec6d922d1b",
    "title": "ViPRA: Video Prediction for Robot Actions",
    "authors": "Sandeep Routray; Hengkai Pan; Unnat Jain; Shikhar Bahl; Deepak Pathak",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "vipra",
    "arxivUrl": null,
    "codeUrls": [
      "https://github.com/sroutray/vipra"
    ],
    "projectUrl": "https://vipra-project.github.io/",
    "venue": "ICLR 2026",
    "paperUrl": "https://iclr.cc/virtual/2026/poster/10006758",
    "pdfUrl": "https://arxiv.org/pdf/2511.07732",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@inproceedings{\nvipra,\ntitle={Vi{PRA}: Video Prediction for Robot Actions},\nauthor={Sandeep Routray and Hengkai Pan and Unnat Jain and Shikhar Bahl and Deepak Pathak},\nbooktitle={NeurIPS Workshop},\nyear={2025},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "联合视频动作建模",
      "泛化与动作对齐",
      "高效推理与实时控制"
    ],
    "architecture": "One Model",
    "predictionParadigm": "联合预测",
    "quadrant": "Q1 · One Model × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-4e998454523b2100fbe1",
    "title": "DrivingWorld: Constructing World Model for Autonomous Driving via Video GPT",
    "authors": "Xiaotao Hu; Mingkai Jia; Xiaoyang Guo; Qian Zhang; Xiao-xiao Long; Wei Yin",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "drivingworld",
    "arxivUrl": null,
    "codeUrls": [
      "https://github.com/YvanYin/DrivingWorld"
    ],
    "projectUrl": "https://huxiaotaostasy.github.io/DrivingWorld/",
    "venue": "ICPR 2026",
    "paperUrl": "https://link.springer.com/chapter/10.1007/978-3-032-31583-0_19",
    "pdfUrl": "https://arxiv.org/pdf/2412.19505",
    "doi": "https://doi.org/10.1007/978-3-032-31583-0_19",
    "publicationYear": 2026,
    "bibtex": "@inproceedings{drivingworld,\n  title={Drivingworld: Constructing world model for autonomous driving via video gpt},\n  author={Hu, Xiaotao and Jia, Mingkai and Guo, Xiaoyang and Zhang, Qian and Long, Xiao-xiao and Yin, Wei},\n  booktitle={ICPR},\n  pages={276--291},\n  year={2026},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "自动驾驶",
      "联合视频动作建模"
    ],
    "architecture": "One Model",
    "predictionParadigm": "联合预测",
    "quadrant": "Q1 · One Model × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-401ac86109e8bd2d687e",
    "title": "DriveWorld-VLA: Unified Latent-Space World Modeling with Vision-Language-Action for Autonomous Driving",
    "authors": "Feiyang jia; Lin Liu; Ziying Song; Caiyan Jia; Hangjun Ye; Xiaoshuai Hao; Long Chen",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "driveworld_vla",
    "arxivUrl": null,
    "codeUrls": [
      "https://github.com/liulin815/DriveWorld-VLA"
    ],
    "projectUrl": null,
    "venue": "ICML 2026",
    "paperUrl": "https://icml.cc/virtual/2026/poster/61191",
    "pdfUrl": "https://arxiv.org/pdf/2602.06521",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@inproceedings{\ndriveworld_vla,\ntitle={DriveWorld-{VLA}: Unified Latent-Space World Modeling with Vision{\\textendash}Language{\\textendash}Action for Autonomous Driving},\nauthor={Feiyang Jia and Lin Liu and Ziying Song and Caiyan Jia and Hangjun Ye and Xiaoshuai Hao and Long Chen},\nbooktitle={ICML},\nyear={2026},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "自动驾驶",
      "潜空间预测与JEPA",
      "策略后训练与WM-RL"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-3e8513f6b1b3fedf96e2",
    "title": "A Careful Examination of Large Behavior Models for Multitask Dexterous Manipulation",
    "authors": "TRI LBM Team; Jose Barreiros; Andrew Beaulieu; Aditya Bhat; Rick Cory; Eric Cousineau; Hongkai Dai; Ching-Hsin Fang; Kunimatsu Hashimoto; Muhammad Zubair Irshad; Masha Itkina; Naveen Kuppuswamy; Kuan-Hui Lee; Katherine Liu; Dale McConachie; Ian McMahon; Haruki Nishimura; Calder Phillips-Grafflin; Charles Richter; Paarth Shah; Krishnan Srinivasan; Blake Wulfe; Chen Xu; Mengchao Zhang; Alex Alspach; Maya Angeles; Kushal Arora; Vitor Campagnolo Guizilini; Alejandro Castro; Dian Chen; Ting-Sheng Chu; Sam Creasey; Sean Curtis; Richard Denitto; Emma Dixon; Eric Dusel; Matthew Ferreira; Aimee Goncalves; Grant Gould; Damrong Guoy; Swati Gupta; Xuchen Han; Kyle Hatch; Brendan Hathaway; Allison Henry; Hillel Hochsztein; Phoebe Horgan; Shun Iwase; Donovon Jackson; Siddharth Karamcheti; Sedrick Keh; Joseph Masterjohn; Jean Mercat; Patrick Miller; Paul Mitiguy; Tony Nguyen; Jeremy Nimmer; Yuki Noguchi; Reko Ong; Aykut Onol; Owen Pfannenstiehl; Richard Poyner; Leticia Priebe Mendes Rocha; Gordon Richardson; Christopher Rodriguez; Derick Seale; Michael Sherman; Mariah Smith-Jones; David Tago; Pavel Tokmakov; Matthew Tran; Basile Van Hoorick; Igor Vasiljevic; Sergey Zakharov; Mark Zolotas; Rares Ambrus; Kerri Fetzer-Borelli; Benjamin Burchfiel; Hadas Kress-Gazit; Siyuan Feng; Stacie Ford; Russ Tedrake",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "LBM",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "Science Robotics",
    "paperUrl": "https://doi.org/10.1126/scirobotics.aea6201",
    "pdfUrl": "https://arxiv.org/pdf/2507.05331",
    "doi": "https://doi.org/10.1126/scirobotics.aea6201",
    "publicationYear": 2026,
    "bibtex": "@article{LBM,\n  title={A careful examination of large behavior models for multitask dexterous manipulation},\n  author={Barreiros, Jose and Beaulieu, Andrew and Bhat, Aditya and Cory, Rick and Cousineau, Eric and Dai, Hongkai and Fang, Ching-Hsin and Hashimoto, Kunimatsu and Irshad, Muhammad Zubair and Itkina, Masha and others},\n  journal={Science Robotics},\n  volume={11},\n  number={113},\n  year={2026},\n}",
    "majorCategory": "VLA",
    "subcategories": [
      "扩散与流匹配VLA"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-3a25df8af09d3c9d3d7a",
    "title": "DriveVLA-W0: World Models Amplify Data Scaling Law in Autonomous Driving",
    "authors": "Yingyan Li; Shuyao Shang; Weisong Liu; Bing Zhan; Haochen Wang; Yuqi Wang; Yuntao Chen; Xiaoman Wang; Yasong An; Chufeng Tang; LU HOU; Lue Fan; Zhaoxiang Zhang",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "drivevla_w0",
    "arxivUrl": null,
    "codeUrls": [
      "https://github.com/BraveGroup/DriveVLA-W0"
    ],
    "projectUrl": null,
    "venue": "ICLR 2026",
    "paperUrl": "https://proceedings.iclr.cc/paper_files/paper/2026/hash/0d70423f59c5fdd24f0dd3fa52e34623-Abstract-Conference.html",
    "pdfUrl": "https://openreview.net/pdf?id=plrGn3RdzN",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@inproceedings{drivevla_w0,\n  title={Drivevla-w0: World models amplify data scaling law in autonomous driving},\n  author={Li, Yingyan and Shang, Shuyao and Liu, Weisong and Zhan, Bing and Wang, Haochen and Wang, Yuqi and Chen, Yuntao and Wang, Xiaoman and An, Yasong and Tang, Chufeng and others},\n  booktitle={ICLR},\n  volume={2026},\n  pages={7890--7911},\n  year={2026}\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "自动驾驶",
      "高效推理与实时控制"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-366f580f3d25b1a75f1c",
    "title": "EgoDex: Learning Dexterous Manipulation from Large-Scale Egocentric Video",
    "authors": "Ryan Hoque; Peide Huang; David J. Yoon; Mouli sivapurapu; Jian Zhang",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "egodex",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "ICLR 2026",
    "paperUrl": "https://iclr.cc/virtual/2026/poster/10010617",
    "pdfUrl": "https://arxiv.org/pdf/2505.11709",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@inproceedings{\negodex,\ntitle={EgoDex: Learning Dexterous Manipulation from Large-Scale Egocentric Video},\nauthor={Ryan Hoque and Peide Huang and David J. Yoon and Mouli sivapurapu and Jian Zhang},\nbooktitle={ICLR},\nyear={2026},\n}",
    "majorCategory": "数据集",
    "subcategories": [
      "人类第一视角数据",
      "灵巧手与抓取数据",
      "三维手部轨迹标注"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-2ea9f2acd9277d70d38e",
    "title": "DrivingGen: A Comprehensive Benchmark for Generative Video World Models in Autonomous Driving",
    "authors": "Yang Zhou; Hao Shao; Letian Wang; Zhuofan Zong; Hongsheng Li; Steven L. Waslander",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "drivinggen",
    "arxivUrl": null,
    "codeUrls": [
      "https://github.com/mr-d-self-driving/drivinggen"
    ],
    "projectUrl": "https://drivinggen-bench.github.io/",
    "venue": "ICLR 2026",
    "paperUrl": "https://openreview.net/forum?id=OrgL5DsU0f",
    "pdfUrl": "https://arxiv.org/pdf/2601.01528",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@inproceedings{drivinggen,\n  title={Drivinggen: A comprehensive benchmark for generative video world models in autonomous driving},\n  author={Zhou, Yang and Shao, Hao and Wang, Letian and Zong, Zhuofan and Li, Hongsheng and Waslander, Steven},\n  booktitle={ICLR},\n  volume={2026},\n  pages={103502--103524},\n  year={2026}\n}",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "世界模型评测基准",
      "自动驾驶基准"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-2597fd7aef4dfb06c60a",
    "title": "WMPO: World Model-based Policy Optimization for Vision-Language-Action Models",
    "authors": "Fangqi Zhu; Zhengyang Yan; Zicong Hong; Quanxin Shou; Xiao Ma; Song Guo",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "wmpo",
    "arxivUrl": null,
    "codeUrls": [
      "https://github.com/WM-PO/WMPO"
    ],
    "projectUrl": "https://wm-po.github.io",
    "venue": "ICLR 2026",
    "paperUrl": "https://iclr.cc/virtual/2026/poster/10007263",
    "pdfUrl": "https://arxiv.org/pdf/2511.09515",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@inproceedings{\nwmpo,\ntitle={{WMPO}: World Model-based Policy Optimization for Vision-Language-Action Models},\nauthor={Fangqi Zhu and YAN Zhengyang and Zicong Hong and Quanxin Shou and Xiao Ma and Song Guo},\nbooktitle={ICLR},\nyear={2026},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "策略后训练与WM-RL"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-1bd6e05d156a43be4c51",
    "title": "DynVLA: Learning World Dynamics for Action Reasoning in Autonomous Driving",
    "authors": "Shuyao Shang; Bing Zhan; Yunfei Yan; Yuqi Wang; Yingyan Li; Yasong An; Xiaoman Wang; Jierui Liu; Lu Hou; Lue Fan; Zhaoxiang Zhang; Tieniu Tan",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "dynvla",
    "arxivUrl": null,
    "codeUrls": [
      "https://github.com/yaoyao-jpg/DynamicsVLA"
    ],
    "projectUrl": "https://yaoyao-jpg.github.io/dynvla/",
    "venue": "ICML 2026",
    "paperUrl": "https://icml.cc/virtual/2026/poster/63736",
    "pdfUrl": "https://arxiv.org/pdf/2603.11041",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@inproceedings{\ndynvla,\ntitle={Dyn{VLA}: Learning World Dynamics for Action Reasoning in Autonomous Driving},\nauthor={Shuyao Shang and Bing Zhan and Yunfei Yan and Yuqi Wang and Yingyan Li and Yasong An and Xiaoman Wang and Jierui Liu and Lu Hou and Lue Fan and Zhaoxiang Zhang and Tieniu Tan},\nbooktitle={ICML},\nyear={2026},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "自动驾驶",
      "潜空间预测与JEPA",
      "视觉规划与IDM"
    ],
    "architecture": "One Model",
    "predictionParadigm": "IDM",
    "quadrant": "Q2 · One Model × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-10cfc9a9bb6b6f4993db",
    "title": "DINOv3",
    "authors": "Oriane Siméoni; Huy V. Vo; Maximilian Seitzer; Federico Baldassarre; Maxime Oquab; Cijo Jose; Vasil Khalidov; Marc Szafraniec; Seungeun Yi; Michaël Ramamonjisoa; Francisco Massa; Daniel Haziza; Luca Wehrstedt; Jianyuan Wang; Timothée Darcet; Théo Moutakanni; Leonel Sentana; Claire Roberts; Andrea Vedaldi; Jamie Tolan; John Brandt; Camille Couprie; Julien Mairal; Hervé Jégou; Patrick Labatut; Piotr Bojanowski",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "dinov3",
    "arxivUrl": null,
    "codeUrls": [
      "https://github.com/facebookresearch/dinov3"
    ],
    "projectUrl": null,
    "venue": "Transactions on Machine Learning Research",
    "paperUrl": "https://openreview.net/forum?id=2NlGyqNjns",
    "pdfUrl": "https://arxiv.org/pdf/2508.10104",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{dinov3,\ntitle={{DINO}v3},\nauthor={Oriane Siméoni and Huy V. Vo and Maximilian Seitzer and Federico Baldassarre and Maxime Oquab and Cijo Jose and Vasil Khalidov and Marc Szafraniec and Seungeun Yi and Michaël Ramamonjisoa and Francisco Massa and Daniel Haziza and Luca Wehrstedt and Jianyuan Wang and Timothée Darcet and Théo Moutakanni and Leonel Sentana and Claire Roberts and Andrea Vedaldi and Jamie Tolan and John Brandt and Camille Couprie and Julien Mairal and Hervé Jégou and Patrick Labatut and Piotr Bojanowski},\njournal={TMLR},\nissn={2835-8856},\nyear={2026},\n}",
    "majorCategory": "WAM Components",
    "subcategories": [
      "视觉编码器与表征"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-0f2536c81a1992e3c3b8",
    "title": "Spatial Forcing: Implicit Spatial Representation Alignment for Vision-Language-Action Model",
    "authors": "Fuhao Li; Wenxuan Song; Han Zhao; Jingbo Wang; Pengxiang Ding; Donglin Wang; Long Zeng; Haoang Li",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "spatialforcing2025",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "ICLR 2026",
    "paperUrl": "https://openreview.net/forum?id=euMVC1DO4k",
    "pdfUrl": "https://openreview.net/pdf?id=euMVC1DO4k",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@inproceedings{spatialforcing2025,\n  title={Spatial forcing: Implicit spatial representation alignment for vision-language-action model},\n  author={Li, Fuhao and Song, Wenxuan and Zhao, Han and Wang, Jingbo and Ding, Pengxiang and Wang, Donglin and Zeng, Long and Li, Haoang},\n  booktitle={ICLR},\n  volume={2026},\n  pages={132324--132345},\n  year={2026}\n}",
    "majorCategory": "VLA",
    "subcategories": [
      "VLA后训练与数据增强",
      "空间感知增强VLA"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-03e87a8030d42fa6e525",
    "title": "Learning Latent Action World Models In The Wild",
    "authors": "Quentin Garrido; Tushar Nagarajan; Basile Terver; Nicolas Ballas; Yann LeCun; Michael Rabbat",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "lawm_wild",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "ICML 2026",
    "paperUrl": "https://icml.cc/virtual/2026/poster/65056",
    "pdfUrl": "https://arxiv.org/pdf/2601.05230",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@inproceedings{\nlawm_wild,\ntitle={Learning Latent Action World Models in the Wild},\nauthor={Quentin Garrido and Tushar Nagarajan and Basile Terver and Nicolas Ballas and Yann LeCun and Michael Rabbat},\nbooktitle={ICML},\nyear={2026},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "潜动作预训练",
      "神经世界模拟器",
      "理论与规划"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-024007cef3657e49b45b",
    "title": "Uncertainty-Aware Robotic World Model Makes Offline Model-Based Reinforcement Learning Work on Real Robots",
    "authors": "Chenhao Li; Andreas Krause; Marco Hutter",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "li2025uncertainty",
    "arxivUrl": null,
    "codeUrls": [
      "https://github.com/leggedrobotics/robotic_world_model"
    ],
    "projectUrl": "https://sites.google.com/view/uncertainty-aware-rwm",
    "venue": "ICLR 2026 Workshop on World Models",
    "paperUrl": "https://openreview.net/pdf/4c7bf7d1b35f9e402b680031bb65e7738103a9bb.pdf",
    "pdfUrl": "https://openreview.net/pdf/4c7bf7d1b35f9e402b680031bb65e7738103a9bb.pdf",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@inproceedings{\nli2025uncertainty,\ntitle={Uncertainty-Aware Robotic World Model Makes Offline Model-Based Reinforcement Learning Work on Real Robots},\nauthor={Chenhao Li and Andreas Krause and Marco Hutter},\nbooktitle={ICLR Workshop},\nyear={2026},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "策略后训练与WM-RL"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2603.20607",
    "title": "Towards Practical World Model-based Reinforcement Learning for Vision-Language-Action Models",
    "authors": "Zhilong Zhang; Haoxiang Ren; Yihao Sun; Yifei Sheng; Haonan Wang; Haoxin Lin; Zhichao Wu; Pierre-Luc Bacon; Yang Yu",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "vla_mbpo",
    "arxivUrl": "https://arxiv.org/abs/2603.20607",
    "codeUrls": [],
    "projectUrl": null,
    "venue": "ICML 2026",
    "paperUrl": "https://arxiv.org/abs/2603.20607",
    "pdfUrl": "https://arxiv.org/pdf/2603.20607",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@inproceedings{\nvla_mbpo,\ntitle={Towards Practical World Model-based Reinforcement Learning for Vision-Language-Action Models},\nauthor={Zhilong Zhang and Haoxiang Ren and Yihao Sun and Yifei Sheng and Haonan Wang and Zhichao Wu and Haoxin Lin and Pierre-Luc Bacon and Yang Yu},\nbooktitle={ICLR Workshop},\nyear={2026},\n\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "策略后训练与WM-RL",
      "三维多视角建模"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2603.08546",
    "title": "Interactive World Simulator for Robot Policy Training and Evaluation",
    "authors": "Yixuan Wang; Rhythm Syed; Fangyu Wu; Mengchao Zhang; Aykut Onol; Jose Barreiros; Hooshang Nayyeri; Tony Dear; Huan Zhang; Yunzhu Li",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "iws",
    "arxivUrl": "https://arxiv.org/abs/2603.08546",
    "codeUrls": [],
    "projectUrl": "https://yixuanwang.me/interactive_world_sim",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2603.08546",
    "pdfUrl": "https://arxiv.org/pdf/2603.08546",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@inproceedings{\niws,\ntitle={Interactive World Simulator for Robot Policy Training and Evaluation},\nauthor={Yixuan Wang and Rhythm Syed and Fangyu Wu and Mengchao Zhang and Aykut Onol and Jose Barreiros and Hooshang Nayyeri and Tony Dear and Huan Zhang and Yunzhu Li},\nbooktitle={RSS Workshop},\nyear={2026},\n}",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "神经世界模拟器",
      "评估指标与协议"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2602.15922",
    "title": "World Action Models are Zero-shot Policies",
    "authors": "Seonghyeon Ye; Yunhao Ge; Kaiyuan Zheng; Shenyuan Gao; Sihyun Yu; George Kurian; Suneel Indupuru; You Liang Tan; Chuning Zhu; Jiannan Xiang; Ayaan Malik; Kyungmin Lee; William Liang; Nadun Ranawaka; Jiasheng Gu; Yinzhen Xu; Guanzhi Wang; Fengyuan Hu; Avnish Narayan; Johan Bjorck; Jing Wang; Gwanghyun Kim; Dantong Niu; Ruijie Zheng; Yuqi Xie; Jimmy Wu; Qi Wang; Ryan Julian; Danfei Xu; Yilun Du; Yevgen Chebotar; Scott Reed; Jan Kautz; Yuke Zhu; Linxi \"Jim\" Fan; Joel Jang",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "dreamzero",
    "arxivUrl": "https://arxiv.org/abs/2602.15922",
    "codeUrls": [],
    "projectUrl": "https://dreamzero0.github.io/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2602.15922",
    "pdfUrl": "https://arxiv.org/pdf/2602.15922",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@inproceedings{\ndreamzero,\ntitle={World Action Models are Zero-shot Policies},\nauthor={Seonghyeon Ye and Yunhao Ge and Kaiyuan Zheng and Shenyuan Gao and Sihyun Yu and George Kurian and Suneel Indupuru and You Liang Tan and Chuning Zhu and Jiannan Xiang and Ayaan Naveed Malik and Kyungmin Lee and William Liang and Nadun Ranawaka Arachchige and Jiasheng Gu and Yinzhen Xu and Guanzhi Wang and Fengyuan Hu and Avnish Narayan and Johan Bjorck and Jing Wang and Gwanghyun Kim and Dantong Niu and Ruijie Zheng and Yuqi Xie and Jimmy Wu and Qi Wang and Danfei Xu and Yilun Du and Ryan Julian and Yevgen Chebotar and Scott Reed and Jan Kautz and Yuke Zhu and Linxi Fan and Joel Jang},\nbooktitle={ICLR Workshop},\nyear={2026},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "视觉规划与IDM",
      "高效推理与实时控制",
      "泛化与动作对齐"
    ],
    "architecture": "One Model",
    "predictionParadigm": "IDM",
    "quadrant": "Q2 · One Model × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2512.08186",
    "title": "Ground Slow, Move Fast: A Dual-System Foundation Model for Generalizable Vision-and-Language Navigation",
    "authors": "Meng Wei; Chenyang Wan; Jiaqi Peng; Xiqian Yu; Yuqiang Yang; Delin Feng; Wenzhe Cai; Chenming Zhu; Tai Wang; Jiangmiao Pang; Xihui Liu",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "internvla_n1",
    "arxivUrl": "https://arxiv.org/abs/2512.08186",
    "codeUrls": [],
    "projectUrl": null,
    "venue": "ICLR 2026",
    "paperUrl": "https://arxiv.org/abs/2512.08186",
    "pdfUrl": "https://openreview.net/pdf/1a25efebc28583e7d38570ab67050516fcc20d8e.pdf",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@inproceedings{internvla_n1,\n  title={Ground slow, move fast: A dual-system foundation model for generalizable vision-language navigation},\n  author={Wei, Meng and Wan, Chenyang and Peng, Peng and Yu, Xiqian and Yang, Yuqiang and Feng, Delin and Cai, Wenzhe and Zhu, Chenming and Wang, Tai and Pang, Jiangmiao and others},\n  booktitle={ICLR},\n  volume={2026},\n  pages={12380--12396},\n  year={2026}\n}",
    "majorCategory": "VLA",
    "subcategories": [
      "分层与双系统VLA"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2512.24766",
    "title": "Dream2Flow: Bridging Video Generation and Open-World Manipulation with 3D Object Flow",
    "authors": "Karthik Dharmarajan; Wenlong Huang; Jiajun Wu; Li Fei-Fei; Ruohan Zhang",
    "affiliations": "Stanford University",
    "contribution": "Dream2Flow converts generated human-interaction videos into 3D object trajectories, then uses domain-specific optimization or reinforcement learning to make a robot realize them. Its central benefit is an object-level interface across embodiments; its reliability still depends on video geometry, tracking, contact assumptions and the downstream controller.",
    "abstract": "",
    "submittedDate": "2025-12-31",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv251224766",
    "arxivUrl": "https://arxiv.org/abs/2512.24766",
    "codeUrls": [],
    "projectUrl": null,
    "venue": "arXiv preprint",
    "paperUrl": "https://arxiv.org/abs/2512.24766",
    "pdfUrl": "https://arxiv.org/pdf/2512.24766",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "视觉规划与IDM",
      "三维多视角建模"
    ],
    "architecture": "待核实",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": null
  },
  {
    "id": "2512.23541",
    "title": "Act2Goal: From World Model To General Goal-conditioned Policy",
    "authors": "Pengfei Zhou; Liliang Chen; Shengcong Chen; Di Chen; Wenzhi Zhao; Rongjun Jin; Guanghui Ren; Jianlan Luo",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2025-12-29",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "act2goal",
    "arxivUrl": "https://arxiv.org/abs/2512.23541",
    "codeUrls": [],
    "projectUrl": "https://act2goal.github.io/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2512.23541",
    "pdfUrl": "https://arxiv.org/pdf/2512.23541",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@article{act2goal,\n      title={Act2Goal: From World Model To General Goal-conditioned Policy}, \n      author={Pengfei Zhou and Liliang Chen and Shengcong Chen and Di Chen and Wenzhi Zhao and Rongjun Jin and Guanghui Ren and Jianlan Luo},\n      year={2025},\n      journal={arXiv:2512.23541},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "视觉规划与IDM",
      "泛化与动作对齐"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2512.16023",
    "title": "CoVAR: Co-generation of Video and Action for Robotic Manipulation via Multi-Modal Diffusion",
    "authors": "Liudi Yang; Yang Bai; George Eskandar; Fengyi Shen; Mohammad Altillawi; Dong Chen; Ziyuan Liu; Abhinav Valada",
    "affiliations": "University of Freiburg; Ludwig Maximilian University of Munich; Munich Center for Machine Learning (MCML); Technical University of Munich; Huawei Heisenberg Research Center (Munich)",
    "contribution": "CoVAR co-generates instruction-conditioned video and robot actions through separate, interacting diffusion transformers. Bridge Attention connects a pretrained video branch to an action branch; a UNet decodes actions, with an additional refinement model for LIBERO90. Reported simulation and UR5 successes support the complete system, while refinement dependence, incomplete evaluation details, and roughly four-second sequence generation limit broader conclusions.",
    "abstract": "",
    "submittedDate": "2025-12-17",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv251216023",
    "arxivUrl": "https://arxiv.org/abs/2512.16023",
    "codeUrls": [],
    "projectUrl": null,
    "venue": "arXiv preprint",
    "paperUrl": "https://arxiv.org/abs/2512.16023",
    "pdfUrl": "https://arxiv.org/pdf/2512.16023",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "联合视频动作建模"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "联合预测",
    "quadrant": "Q3 · Dual-system × 联合预测",
    "classificationStatus": null
  },
  {
    "id": "2512.15692",
    "title": "mimic-video: Video-Action Models for Generalizable Robot Control Beyond VLAs",
    "authors": "Jonas Pai; Liam Achenbach; Victoriano Montesinos; Benedek Forrai; Oier Mees; Elvis Nava",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2025-12-17",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "mimic_video",
    "arxivUrl": "https://arxiv.org/abs/2512.15692",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2512.15692",
    "pdfUrl": "https://arxiv.org/pdf/2512.15692",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@article{mimic_video,\n      title={mimic-video: Video-Action Models for Generalizable Robot Control Beyond VLAs}, \n      author={Jonas Pai and Liam Achenbach and Victoriano Montesinos and Benedek Forrai and Oier Mees and Elvis Nava},\n      year={2025},\n      journal={arXiv:2512.15692},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "视觉规划与IDM",
      "高效推理与实时控制"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2512.13644",
    "title": "World Models for Learning Dexterous Hand-Object Interactions from Human Videos",
    "authors": "Raktim Gautam Goswami; Amir Bar; David Fan; Tsung-Yen Yang; Gaoyue Zhou; Prashanth Krishnamurthy; Michael Rabbat; Farshad Khorrami; Yann LeCun",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2025-12-15",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "dex_wm",
    "arxivUrl": "https://arxiv.org/abs/2512.13644",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2512.13644",
    "pdfUrl": "https://arxiv.org/pdf/2512.13644",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@article{dex_wm,\n      title={World Models for Learning Dexterous Hand-Object Interactions from Human Videos}, \n      author={Raktim Gautam Goswami and Amir Bar and David Fan and Tsung-Yen Yang and Gaoyue Zhou and Prashanth Krishnamurthy and Michael Rabbat and Farshad Khorrami and Yann LeCun},\n      year={2025},\n      journal={arXiv:2512.13644},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "潜空间预测与JEPA",
      "泛化与动作对齐",
      "灵巧操作"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2512.10958",
    "title": "WorldLens: Full-Spectrum Evaluations of Driving World Models in Real World",
    "authors": "Ao Liang; Lingdong Kong; Tianyi Yan; Hongsi Liu; Wesley Yang; Ziqi Huang; Wei Yin; Jialong Zuo; Yixuan Hu; Dekai Zhu; Dongyue Lu; Youquan Liu; Guangfeng Jiang; Linfeng Li; Xiangtai Li; Long Zhuo; Lai Xing Ng; Benoit R. Cottereau; Changxin Gao; Liang Pan; Wei Tsang Ooi; Ziwei Liu",
    "affiliations": "",
    "contribution": "WorldLens evaluates driving video generators through appearance, reconstructability, planner behavior, perception and human judgment. Its strongest lesson is that favorable image metrics coexist with poor closed-loop route completion. A separate LoRA-trained critic learns score-and-rationale outputs from human annotations; its generalization evidence remains qualitative.",
    "abstract": "",
    "submittedDate": "2025-12-11",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv251210958",
    "arxivUrl": "https://arxiv.org/abs/2512.10958",
    "codeUrls": [],
    "projectUrl": null,
    "venue": "CVPR 2026",
    "paperUrl": "https://arxiv.org/abs/2512.10958",
    "pdfUrl": "https://arxiv.org/pdf/2512.10958",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "世界模型评测基准",
      "自动驾驶基准"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": null
  },
  {
    "id": "2512.06963",
    "title": "VideoVLA: Video Generators Can Be Generalizable Robot Manipulators",
    "authors": "Yichao Shen; Fangyun Wei; Zhiying Du; Yaobo Liang; Yan Lu; Jiaolong Yang; Nanning Zheng; Baining Guo",
    "affiliations": "IAIR, Xi’an Jiaotong University; Microsoft Research Asia; Fudan University",
    "contribution": "VideoVLA adapts CogVideoX-5B into a robot policy that jointly denoises future video latents and executable action chunks, conditioned on language and the current image. Its clearest gains concern novel objects and skills transferred between embodiments. Video supervision matters strongly in the reported ablations, but imagined success exceeds executed success and deployment remains slow.",
    "abstract": "",
    "submittedDate": "2025-12-07",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv251206963",
    "arxivUrl": "https://arxiv.org/abs/2512.06963",
    "codeUrls": [],
    "projectUrl": null,
    "venue": "NeurIPS 2025",
    "paperUrl": "https://arxiv.org/abs/2512.06963",
    "pdfUrl": "https://arxiv.org/pdf/2512.06963",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "联合视频动作建模"
    ],
    "architecture": "One Model",
    "predictionParadigm": "联合预测",
    "quadrant": "Q1 · One Model × 联合预测",
    "classificationStatus": null
  },
  {
    "id": "2512.06628",
    "title": "MIND-V: Hierarchical World Model for Long-Horizon Robotic Manipulation with RL-based Physical Alignment",
    "authors": "Ruicheng Zhang; Mingyang Zhang; Jun Zhou; Xiaofan Liu; Zunnan Xu; Zhizhou Zhong; Puxin Yan; Haocheng Luo; Xiu Li",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2025-12-07",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "mind_v",
    "arxivUrl": "https://arxiv.org/abs/2512.06628",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2512.06628",
    "pdfUrl": "https://arxiv.org/pdf/2512.06628",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@article{mind_v,\n      title={MIND-V: Hierarchical World Model for Long-Horizon Robotic Manipulation with RL-based Physical Alignment}, \n      author={Ruicheng Zhang and Mingyang Zhang and Jun Zhou and Zhangrui Guo and Zunnan Xu and Xiaofan Liu and Zhizhou Zhong and Puxin Yan and Haocheng Luo and Xiu Li},\n      year={2025},\n      journal={arXiv:2512.06628},\n}",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "神经世界模拟器",
      "训练优化与蒸馏"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2512.04441",
    "title": "MindDrive: An All-in-One Framework Bridging World Models and Vision-Language Model for End-to-End Autonomous Driving",
    "authors": "Bin Sun; Yaoguang Cao; Yan Wang; Rui Wang; Jiachen Shang; Xiejie Feng; Jiayi Lu; Jia Shi; Shichun Yang; Xiaoyu Yan; Ziying Song",
    "affiliations": "School of Transportation Science and Engineering, Beihang University; State Key Laboratory of Intelligent Transportation System, Beihang University; Hangzhou International Innovation Institute, Beihang University; Contemporary Amperex Technology Co., Limited (CATL); Research Institute of Aero-Engine, Beihang University; School of Computer Science and Technology, Beijing Jiaotong University; China Automotive Engineering Research Institute Co., Ltd.",
    "contribution": "MindDrive couples action-conditioned BEV scene prediction with anchor refinement and a VLM trajectory scorer. Its main contribution is the connection between future-aware candidate generation and selection. NAVSIM scores support planning gains, while incomplete training specifications and inconsistent result reporting limit reproducibility.",
    "abstract": "",
    "submittedDate": "2025-12-04",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv251204441",
    "arxivUrl": "https://arxiv.org/abs/2512.04441",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2512.04441",
    "pdfUrl": "https://arxiv.org/pdf/2512.04441",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "自动驾驶"
    ],
    "architecture": "待核实",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": null
  },
  {
    "id": "2512.03044",
    "title": "Video2Act: A Dual-System Video Diffusion Policy with Robotic Spatio-Motional Modeling",
    "authors": "Yueru Jia; Jiaming Liu; Shengbang Liu; Rui Zhou; Wanhe Yu; Yuyang Yan; Xiaowei Chi; Yandong Guo; Boxin Shi; Shanghang Zhang",
    "affiliations": "State Key Laboratory of Multimedia Information Processing, School of Computer Science, Peking University; AI²Robotics; Hong Kong University of Science and Technology",
    "contribution": "Video2Act turns observed video into conditioning for a separate robot action policy. Sobel filtering emphasizes spatial structure in video-model features, while temporal Fourier filtering emphasizes motion. A slow video network refreshes these representations for a faster diffusion action head. The strongest evidence is improved executed manipulation on RoboTwin and a small real-robot evaluation; the speed claims require separating action-chunk throughput from fresh-feedback control.",
    "abstract": "",
    "submittedDate": "2025-12-02",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv251203044",
    "arxivUrl": "https://arxiv.org/abs/2512.03044",
    "codeUrls": [],
    "projectUrl": null,
    "venue": "arXiv preprint",
    "paperUrl": "https://arxiv.org/abs/2512.03044",
    "pdfUrl": "https://arxiv.org/pdf/2512.03044",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "",
    "majorCategory": "VLA",
    "subcategories": [
      "分层与双系统VLA",
      "扩散与流匹配VLA"
    ],
    "architecture": null,
    "predictionParadigm": null,
    "quadrant": null,
    "classificationStatus": null
  },
  {
    "id": "2511.23034",
    "title": "LatBot: Distilling Universal Latent Actions for Vision-Language-Action Models",
    "authors": "Zuolei Li; Xingyu Gao; Xiaofan Wang; Jianlong Fu",
    "affiliations": "Institute of Microelectronics, Chinese Academy of Sciences; University of Chinese Academy of Sciences; Microsoft Research",
    "contribution": "LatBot learns scene and motion latents from language-conditioned robot and human manipulation videos, supervises them through future-image and physical-action decoding, then distills them into a VLA student. An action expert converts the student's features into executable commands. The strongest evidence concerns downstream manipulation success; universal physical understanding and preserved reasoning remain broader interpretations of these results (e03–e06, e12–e16).",
    "abstract": "",
    "submittedDate": "2025-11-28",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv251123034",
    "arxivUrl": "https://arxiv.org/abs/2511.23034",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2511.23034",
    "pdfUrl": "https://arxiv.org/pdf/2511.23034",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "",
    "majorCategory": "VLA",
    "subcategories": [
      "潜动作预训练"
    ],
    "architecture": null,
    "predictionParadigm": null,
    "quadrant": null,
    "classificationStatus": null
  },
  {
    "id": "2511.19861",
    "title": "GigaWorld-0: World Models as Data Engine to Empower Embodied AI",
    "authors": "GigaWorld Team; Angen Ye; Boyuan Wang; Chaojun Ni; Guan Huang; Guosheng Zhao; Haoyun Li; Jiagang Zhu; Kerui Li; Mengyuan Xu; Qiuping Deng; Siting Wang; Wenkang Qin; Xinze Chen; Xiaofeng Wang; Yankai Wang; Yu Cao; Yifan Chang; Yuan Xu; Yun Ye; Yang Wang; Yukun Zhou; Zhengyuan Zhang; Zhehao Dong; Zheng Zhu",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2025-11-25",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "gigaworld0",
    "arxivUrl": "https://arxiv.org/abs/2511.19861",
    "codeUrls": [
      "https://github.com/open-gigaai/giga-world-0"
    ],
    "projectUrl": "https://giga-world-0.github.io/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2511.19861",
    "pdfUrl": "https://arxiv.org/pdf/2511.19861",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@article{gigaworld0,\n      title={GigaWorld-0: World Models as Data Engine to Empower Embodied AI}, \n      author={GigaWorld Team and Angen Ye and Boyuan Wang and Chaojun Ni and Guan Huang and Guosheng Zhao and Haoyun Li and Jiagang Zhu and Kerui Li and Mengyuan Xu and Qiuping Deng and Siting Wang and Wenkang Qin and Xinze Chen and Xiaofeng Wang and Yankai Wang and Yu Cao and Yifan Chang and Yuan Xu and Yun Ye and Yang Wang and Yukun Zhou and Zhengyuan Zhang and Zhehao Dong and Zheng Zhu},\n      year={2025},\n      journal={arXiv:2511.19861},\n}",
    "majorCategory": "数据集",
    "subcategories": [
      "合成数据与数据生成",
      "三维场景数据生成",
      "机器人交互数据"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2511.19584",
    "title": "Learning Massively Multitask World Models for Continuous Control",
    "authors": "Nicklas Hansen; Hao Su; Xiaolong Wang",
    "affiliations": "University of California San Diego",
    "contribution": "Newt extends TD-MPC2 into a language-conditioned agent that learns latent dynamics from demonstrations and online interaction across MMBench. Its practical recipe pretrains the world model and policy prior, retains action supervision, and uses the model to plan. It improves over the evaluated multitask baselines, but specialist policies remain stronger and long open-loop execution is uneven. The evidence concerns simulated, predominantly state-based control.",
    "abstract": "",
    "submittedDate": "2025-11-24",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv251119584",
    "arxivUrl": "https://arxiv.org/abs/2511.19584",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2511.19584",
    "pdfUrl": "https://arxiv.org/pdf/2511.19584",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "策略后训练与WM-RL"
    ],
    "architecture": "待核实",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": null
  },
  {
    "id": "2511.17502",
    "title": "RynnVLA-002: A Unified Vision-Language-Action and World Model",
    "authors": "Jun Cen; Siteng Huang; Yuqian Yuan; Kehan Li; Hangjie Yuan; Chaohui Yu; Bohan Hou; Yuming Jiang; Jiayan Guo; Xin Li; Hao Luo; Fan Wang; Deli Zhao; Hao Chen",
    "affiliations": "DAMO Academy, Alibaba Group; Hupan Lab; Zhejiang University",
    "contribution": "RynnVLA-002 finetunes a shared Chameleon backbone for action prediction and action-conditioned image prediction, then adds a parallel continuous-action head. Its strongest evidence is mutual training benefit: better executed policies and better held-out visual predictions. Policy inference uses no imagined-image rollout. The reported 97.4% LIBERO average is competitive, while physical evidence is limited to SO100 pick-and-place.",
    "abstract": "",
    "submittedDate": "2025-11-21",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv251117502",
    "arxivUrl": "https://arxiv.org/abs/2511.17502",
    "codeUrls": [],
    "projectUrl": null,
    "venue": "arXiv preprint",
    "paperUrl": "https://arxiv.org/abs/2511.17502",
    "pdfUrl": "https://arxiv.org/pdf/2511.17502",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "联合视频动作建模"
    ],
    "architecture": null,
    "predictionParadigm": null,
    "quadrant": null,
    "classificationStatus": null
  },
  {
    "id": "2511.16661",
    "title": "Dexterity from Smart Lenses: Multi-Fingered Robot Manipulation with In-the-Wild Human Demonstrations",
    "authors": "Irmak Guzey; Haozhi Qi; Julen Urain; Changhao Wang; Jessica Yin; Krishna Bodduluri; Mike Lambeta; Lerrel Pinto; Akshara Rai; Jitendra Malik; Tingfan Wu; Akash Sharma; Homanga Bharadhwaj",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2025-11-20",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "Guzey2025DexteritySmart",
    "arxivUrl": "https://arxiv.org/abs/2511.16661",
    "codeUrls": [
      "https://github.com/facebookresearch/AINA/tree/main"
    ],
    "projectUrl": "https://aina-robot.github.io",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2511.16661",
    "pdfUrl": "https://arxiv.org/pdf/2511.16661",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@article{Guzey2025DexteritySmart,\n      title={Dexterity from Smart Lenses: Multi-Fingered Robot Manipulation with In-the-Wild Human Demonstrations}, \n      author={Irmak Guzey and Haozhi Qi and Julen Urain and Changhao Wang and Jessica Yin and Krishna Bodduluri and Mike Lambeta and Lerrel Pinto and Akshara Rai and Jitendra Malik and Tingfan Wu and Akash Sharma and Homanga Bharadhwaj},\n      year={2025},\n      journal={arXiv:2511.16661},\n}",
    "majorCategory": "Related resources",
    "subcategories": [
      "动作策略基础",
      "数据集与数据采集"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2511.14759",
    "title": "π^*_0.6: a VLA That Learns From Experience",
    "authors": "Physical Intelligence; Ali Amin; Raichelle Aniceto; Ashwin Balakrishna; Kevin Black; Ken Conley; Grace Connors; James Darpinian; Karan Dhabalia; Jared DiCarlo; Danny Driess; Michael Equi; Adnan Esmail; Yunhao Fang; Chelsea Finn; Catherine Glossop; Thomas Godden; Ivan Goryachev; Lachy Groom; Hunter Hancock; Karol Hausman; Gashon Hussein; Brian Ichter; Szymon Jakubczak; Rowan Jen; Tim Jones; Ben Katz; Liyiming Ke; Chandra Kuchi; Marinda Lamb; Devin LeBlanc; Sergey Levine; Adrian Li-Bell; Yao Lu; Vishnu Mano; Mohith Mothukuri; Suraj Nair; Karl Pertsch; Allen Z. Ren; Charvi Sharma; Lucy Xiaoyang Shi; Laura Smith; Jost Tobias Springenberg; Kyle Stachowicz; Will Stoeckle; Alex Swerdlow; James Tanner; Marcel Torne; Quan Vuong; Anna Walling; Haohuan Wang; Blake Williams; Sukwon Yoo; Lili Yu; Ury Zhilinsky; Zhiyuan Zhou",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2025-11-18",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "pi0_6",
    "arxivUrl": "https://arxiv.org/abs/2511.14759",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2511.14759",
    "pdfUrl": "https://arxiv.org/pdf/2511.14759",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@article{pi0_6,\n  title={{$\\pi_{0.6}$: a VLA That Learns From Experience}},\n  author={Intelligence, Physical and Amin, Ali and Aniceto, Raichelle and Balakrishna, Ashwin and Black, Kevin and Conley, Ken and Connors, Grace and Darpinian, James and Dhabalia, Karan and DiCarlo, Jared and others},\n  journal={arXiv:2511.14759},\n  year={2025}\n}",
    "majorCategory": "VLA",
    "subcategories": [
      "VLA后训练与数据增强",
      "扩散与流匹配VLA"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2511.11520",
    "title": "Scalable Policy Evaluation with Video World Models",
    "authors": "Wei-Cheng Tseng; Jinwei Gu; Qinsheng Zhang; Hanzi Mao; Ming-Yu Liu; Florian Shkurti; Lin Yen-Chen",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2025-11-14",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "spe_vwm",
    "arxivUrl": "https://arxiv.org/abs/2511.11520",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2511.11520",
    "pdfUrl": "https://arxiv.org/pdf/2511.11520",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@article{spe_vwm,\n      title={Scalable Policy Evaluation with Video World Models}, \n      author={Wei-Cheng Tseng and Jinwei Gu and Qinsheng Zhang and Hanzi Mao and Ming-Yu Liu and Florian Shkurti and Lin Yen-Chen},\n      year={2025},\n      journal={arXiv:2511.11520},\n}",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "神经策略评测环境",
      "机器人操作基准",
      "仿真到真实评测"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2511.09057",
    "title": "PAN: A World Model for General, Interactable, and Long-Horizon World Simulation",
    "authors": "PAN Team; Jiannan Xiang; Yi Gu; Zihan Liu; Zeyu Feng; Qiyue Gao; Yiyan Hu; Benhao Huang; Guangyi Liu; Yichi Yang; Kun Zhou; Davit Abrahamyan; Arif Ahmad; Ganesh Bannur; Junrong Chen; Kimi Chen; Mingkai Deng; Ruobing Han; Xinqi Huang; Haoqiang Kang; Zheqi Liu; Enze Ma; Hector Ren; Yashowardhan Shinde; Rohan Shingre; Ramsundar Tanikella; Kaiming Tao; Dequan Yang; Xinle Yu; Cong Zeng; Binglin Zhou; Zhengzhong Liu; Zhiting Hu; Eric P. Xing",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2025-11-12",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "pan",
    "arxivUrl": "https://arxiv.org/abs/2511.09057",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2511.09057",
    "pdfUrl": "https://arxiv.org/pdf/2511.09057",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@article{pan,\n      title={Pan: A world model for general, interactable, and long-horizon world simulation}, \n      author={Xiang, Jiannan and Gu, Yi and Liu, Zihan and Feng, Zeyu and Gao, Qiyue and Hu, Yiyan and Huang, Benhao and Liu, Guangyi and Yang, Yichi and Zhou, Kun and others},\n      year={2025},\n      journal={arXiv:2511.09057},\n}",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "神经世界模拟器"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2510.27607",
    "title": "Dual-Stream Diffusion for World-Model Augmented Vision-Language-Action Model",
    "authors": "John Won; Kyungmin Lee; Huiwon Jang; Dongyoung Kim; Jinwoo Shin",
    "affiliations": "Kim Jaechul Graduate School of AI, Korea Advanced Institute of Technology, Seoul, Republic of Korea; RLWRLD, Seoul, Republic of Korea",
    "contribution": "DUST augments a frozen vision-language backbone with jointly denoised actions and future visual embeddings. Separate streams exchange information through attention; independent noise levels and unequal sampling budgets accommodate their different dynamics. Controlled comparisons support improved manipulation, but do not establish general causal world understanding.",
    "abstract": "",
    "submittedDate": "2025-10-31",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv251027607",
    "arxivUrl": "https://arxiv.org/abs/2510.27607",
    "codeUrls": [],
    "projectUrl": null,
    "venue": "ICML 2026",
    "paperUrl": "https://arxiv.org/abs/2510.27607",
    "pdfUrl": "https://arxiv.org/pdf/2510.27607",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "联合视频动作建模",
      "潜空间预测与JEPA"
    ],
    "architecture": null,
    "predictionParadigm": null,
    "quadrant": null,
    "classificationStatus": null
  },
  {
    "id": "2510.25889",
    "title": "π_RL: Online RL Fine-tuning for Flow-based Vision-Language-Action Models",
    "authors": "Kang Chen; Zhihao Liu; Tonghe Zhang; Zhen Guo; Si Xu; Hao Lin; Hongzhi Zang; Xiang Li; Quanlu Zhang; Zhaofei Yu; Guoliang Fan; Tiejun Huang; Yu Wang; Chao Yu",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2025-10-29",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "pi_rl",
    "arxivUrl": "https://arxiv.org/abs/2510.25889",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2510.25889",
    "pdfUrl": "https://arxiv.org/pdf/2510.25889",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@article{pi_rl,\n      title={{$\\pi_\\texttt{RL}$: Online RL Fine-tuning for Flow-based Vision-Language-Action Models}}, \n      author={Kang Chen and Zhihao Liu and Tonghe Zhang and Zhen Guo and Si Xu and Hao Lin and Hongzhi Zang and Xiang Li and Quanlu Zhang and Zhaofei Yu and Guoliang Fan and Tiejun Huang and Yu Wang and Chao Yu},\n      year={2025},\n      journal={arXiv:2510.25889},\n}",
    "majorCategory": "VLA",
    "subcategories": [
      "VLA后训练与数据增强",
      "扩散与流匹配VLA"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2511.00062",
    "title": "World Simulation with Video Foundation Models for Physical AI",
    "authors": "NVIDIA; :; Arslan Ali; Junjie Bai; Maciej Bala; Yogesh Balaji; Aaron Blakeman; Tiffany Cai; Jiaxin Cao; Tianshi Cao; Elizabeth Cha; Yu-Wei Chao; Prithvijit Chattopadhyay; Mike Chen; Yongxin Chen; Yu Chen; Shuai Cheng; Yin Cui; Jenna Diamond; Yifan Ding; Jiaojiao Fan; Linxi Fan; Liang Feng; Francesco Ferroni; Sanja Fidler; Xiao Fu; Ruiyuan Gao; Yunhao Ge; Jinwei Gu; Aryaman Gupta; Siddharth Gururani; Imad El Hanafi; Ali Hassani; Zekun Hao; Jacob Huffman; Joel Jang; Pooya Jannaty; Jan Kautz; Grace Lam; Xuan Li; Zhaoshuo Li; Maosheng Liao; Chen-Hsuan Lin; Tsung-Yi Lin; Yen-Chen Lin; Huan Ling; Ming-Yu Liu; Xian Liu; Yifan Lu; Alice Luo; Qianli Ma; Hanzi Mao; Kaichun Mo; Seungjun Nah; Yashraj Narang; Abhijeet Panaskar; Lindsey Pavao; Trung Pham; Morteza Ramezanali; Fitsum Reda; Scott Reed; Xuanchi Ren; Haonan Shao; Yue Shen; Stella Shi; Shuran Song; Bartosz Stefaniak; Shangkun Sun; Shitao Tang; Sameena Tasmeen; Lyne Tchapmi; Wei-Cheng Tseng; Jibin Varghese; Andrew Z. Wang; Hao Wang; Haoxiang Wang; Heng Wang; Ting-Chun Wang; Fangyin Wei; Jiashu Xu; Dinghao Yang; Xiaodong Yang; Haotian Ye; Seonghyeon Ye; Xiaohui Zeng; Jing Zhang; Qinsheng Zhang; Kaiwen Zheng; Andrew Zhu; Yuke Zhu",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2025-10-28",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "cosmos",
    "arxivUrl": "https://arxiv.org/abs/2511.00062",
    "codeUrls": [
      "https://github.com/nvidia-cosmos/cosmos-predict2.5"
    ],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2511.00062",
    "pdfUrl": "https://arxiv.org/pdf/2511.00062",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@article{cosmos,\n      title={World Simulation with Video Foundation Models for Physical AI}, \n      author={NVIDIA and : and Arslan Ali and Junjie Bai and Maciej Bala and Yogesh Balaji and Aaron Blakeman and Tiffany Cai and Jiaxin Cao and Tianshi Cao and Elizabeth Cha and Yu-Wei Chao and Prithvijit Chattopadhyay and Mike Chen and Yongxin Chen and Yu Chen and Shuai Cheng and Yin Cui and Jenna Diamond and Yifan Ding and Jiaojiao Fan and Linxi Fan and Liang Feng and Francesco Ferroni and Sanja Fidler and Xiao Fu and Ruiyuan Gao and Yunhao Ge and Jinwei Gu and Aryaman Gupta and Siddharth Gururani and Imad El Hanafi and Ali Hassani and Zekun Hao and Jacob Huffman and Joel Jang and Pooya Jannaty and Jan Kautz and Grace Lam and Xuan Li and Zhaoshuo Li and Maosheng Liao and Chen-Hsuan Lin and Tsung-Yi Lin and Yen-Chen Lin and Huan Ling and Ming-Yu Liu and Xian Liu and Yifan Lu and Alice Luo and Qianli Ma and Hanzi Mao and Kaichun Mo and Seungjun Nah and Yashraj Narang and Abhijeet Panaskar and Lindsey Pavao and Trung Pham and Morteza Ramezanali and Fitsum Reda and Scott Reed and Xuanchi Ren and Haonan Shao and Yue Shen and Stella Shi and Shuran Song and Bartosz Stefaniak and Shangkun Sun and Shitao Tang and Sameena Tasmeen and Lyne Tchapmi and Wei-Cheng Tseng and Jibin Varghese and Andrew Z. Wang and Hao Wang and Haoxiang Wang and Heng Wang and Ting-Chun Wang and Fangyin Wei and Jiashu Xu and Dinghao Yang and Xiaodong Yang and Haotian Ye and Seonghyeon Ye and Xiaohui Zeng and Jing Zhang and Qinsheng Zhang and Kaiwen Zheng and Andrew Zhu and Yuke Zhu},\n      year={2025},\n      journal={arXiv:2511.00062},\n}",
    "majorCategory": "WAM Components",
    "subcategories": [
      "Video & world prediction backbones",
      "视频生成Backbone"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2510.19430",
    "title": "GigaBrain-0: A World Model-Powered Vision-Language-Action Model",
    "authors": "GigaBrain Team; Angen Ye; Boyuan Wang; Chaojun Ni; Guan Huang; Guosheng Zhao; Haoyun Li; Jie Li; Jiagang Zhu; Lv Feng; Peng Li; Qiuping Deng; Runqi Ouyang; Wenkang Qin; Xinze Chen; Xiaofeng Wang; Yang Wang; Yifan Li; Yilong Li; Yiran Ding; Yuan Xu; Yun Ye; Yukun Zhou; Zhehao Dong; Zhenan Wang; Zhichao Liu; Zheng Zhu",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2025-10-22",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "gigabrain0",
    "arxivUrl": "https://arxiv.org/abs/2510.19430",
    "codeUrls": [
      "https://github.com/open-gigaai/giga-brain-0"
    ],
    "projectUrl": "https://gigabrain0.github.io/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2510.19430",
    "pdfUrl": "https://arxiv.org/pdf/2510.19430",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@article{gigabrain0,\n      title={GigaBrain-0: A World Model-Powered Vision-Language-Action Model}, \n      author={GigaBrain Team and Angen Ye and Boyuan Wang and Chaojun Ni and Guan Huang and Guosheng Zhao and Haoyun Li and Jie Li and Jiagang Zhu and Lv Feng and Peng Li and Qiuping Deng and Runqi Ouyang and Wenkang Qin and Xinze Chen and Xiaofeng Wang and Yang Wang and Yifan Li and Yilong Li and Yiran Ding and Yuan Xu and Yun Ye and Yukun Zhou and Zhehao Dong and Zhenan Wang and Zhichao Liu and Zheng Zhu},\n      year={2025},\n      journal={arXiv:2510.19430},\n}",
    "majorCategory": "VLA",
    "subcategories": [
      "VLA后训练与数据增强",
      "分层与双系统VLA"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2510.13626",
    "title": "LIBERO-Plus: In-depth Robustness Analysis of Vision-Language-Action Models",
    "authors": "Senyu Fei; Siyin Wang; Junhao Shi; Zihao Dai; Jikun Cai; Pengfang Qian; Li Ji; Xinzhe He; Shiduo Zhang; Zhaoye Fei; Jinlan Fu; Jingjing Gong; Xipeng Qiu",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2025-10-15",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "Fei2025LIBEROPlusIndepth",
    "arxivUrl": "https://arxiv.org/abs/2510.13626",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2510.13626",
    "pdfUrl": "https://arxiv.org/pdf/2510.13626",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@article{Fei2025LIBEROPlusIndepth,\n      title={LIBERO-Plus: In-depth Robustness Analysis of Vision-Language-Action Models}, \n      author={Senyu Fei and Siyin Wang and Junhao Shi and Zihao Dai and Jikun Cai and Pengfang Qian and Li Ji and Xinzhe He and Shiduo Zhang and Zhaoye Fei and Jinlan Fu and Jingjing Gong and Xipeng Qiu},\n      year={2025},\n      journal={arXiv:2510.13626},\n}",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "机器人操作基准",
      "鲁棒性与泛化评测"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2510.11682",
    "title": "Ego-Vision World Model for Humanoid Contact Planning",
    "authors": "Hang Liu; Yuman Gao; Sangli Teng; Yufeng Chi; Yakun Sophia Shao; Zhongyu Li; Maani Ghaffari; Koushil Sreenath",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2025-10-13",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "ego_vision_wm",
    "arxivUrl": "https://arxiv.org/abs/2510.11682",
    "codeUrls": [
      "https://github.com/HybridRobotics/Ego-VCP"
    ],
    "projectUrl": "https://ego-vcp.github.io/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2510.11682",
    "pdfUrl": "https://arxiv.org/pdf/2510.11682",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@article{ego_vision_wm,\n      title={Ego-Vision World Model for Humanoid Contact Planning}, \n      author={Hang Liu and Yuman Gao and Sangli Teng and Yufeng Chi and Yakun Sophia Shao and Zhongyu Li and Maani Ghaffari and Koushil Sreenath},\n      year={2025},\n      journal={arXiv:2510.11682},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "三维多视角建模",
      "视觉规划与IDM"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2510.07092",
    "title": "Generative World Modelling for Humanoids: 1X World Model Challenge Technical Report",
    "authors": "Riccardo Mereu; Aidan Scannell; Yuxin Hou; Yi Zhao; Aditya Jitta; Antonio Dominguez; Luigi Acerbi; Amos Storkey; Paul Chang",
    "affiliations": "Aalto University; University of Edinburgh; Deep Render; DataCrunch; University of Helsinki",
    "contribution": "Team Revontuli uses two separate, state-conditioned predictors: a LoRA-adapted Wan video model for future RGB frames and a spatio-temporal Transformer for future discrete tokens. Both lead the reported challenge leaderboard, but their scores measure conditional prediction rather than executed humanoid control.",
    "abstract": "",
    "submittedDate": "2025-10-08",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv251007092",
    "arxivUrl": "https://arxiv.org/abs/2510.07092",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2510.07092",
    "pdfUrl": "https://arxiv.org/pdf/2510.07092",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "神经世界模拟器"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": null
  },
  {
    "id": "2509.22199",
    "title": "MimicDreamer: Aligning Human and Robot Demonstrations for Scalable VLA Training",
    "authors": "Haoyun Li; Ivan Zhang; Runqi Ouyang; Xiaofeng Wang; Zheng Zhu; Zhiqin Yang; Zhentao Zhang; Boyuan Wang; Chaojun Ni; Wenkang Qin; Xinze Chen; Yun Ye; Guan Huang; Zhenbo Song; Xingang Wang",
    "affiliations": "GigaAI; CASIA; NJUST; Tsinghua University",
    "contribution": "MimicDreamer converts human demonstrations into robot-looking videos paired with retargeted actions, then post-trains a separate π0 policy. The contribution is training-data alignment across viewpoint, embodiment and appearance. Physical-task results improve with added synthetic demonstrations, but real-robot supervision remains part of the evaluated pipeline and several headline summaries conflict with the tables.",
    "abstract": "",
    "submittedDate": "2025-09-26",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv250922199",
    "arxivUrl": "https://arxiv.org/abs/2509.22199",
    "codeUrls": [],
    "projectUrl": null,
    "venue": "arXiv preprint",
    "paperUrl": "https://arxiv.org/abs/2509.22199",
    "pdfUrl": "https://arxiv.org/pdf/2509.22199",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "",
    "majorCategory": "数据集",
    "subcategories": [
      "合成数据与数据生成",
      "伪动作标注"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": null
  },
  {
    "id": "2509.21797",
    "title": "MoWM: Mixture-of-World-Models for Embodied Planning via Latent-to-Pixel Feature Modulation",
    "authors": "Yangcheng Yu; Xin Jin; Yu Shang; Xin Zhang; Haisheng Su; Wei Wu; Yong Li",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2025-09-26",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "mowm",
    "arxivUrl": "https://arxiv.org/abs/2509.21797",
    "codeUrls": [
      "https://github.com/tsinghua-fib-lab/MoWM"
    ],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2509.21797",
    "pdfUrl": "https://arxiv.org/pdf/2509.21797",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@article{mowm,\n      title={MoWM: Mixture-of-World-Models for Embodied Planning via Latent-to-Pixel Feature Modulation}, \n      author={Yangcheng Yu and Xin Jin and Yu Shang and Xin Zhang and Haisheng Su and Wei Wu and Yong Li},\n      year={2025},\n      journal={arXiv:2509.21797},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "视觉规划与IDM",
      "潜空间预测与JEPA"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2509.21790",
    "title": "LongScape: Advancing Long-Horizon Embodied World Models with Context-Aware MoE",
    "authors": "Yu Shang; Lei Jin; Yiding Ma; Xin Zhang; Chen Gao; Wei Wu; Yong Li",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2025-09-26",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "longscape",
    "arxivUrl": "https://arxiv.org/abs/2509.21790",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2509.21790",
    "pdfUrl": "https://arxiv.org/pdf/2509.21790",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@article{longscape,\n      title={LongScape: Advancing Long-Horizon Embodied World Models with Context-Aware MoE}, \n      author={Yu Shang and Lei Jin and Yiding Ma and Xin Zhang and Chen Gao and Wei Wu and Yong Li},\n      year={2025},\n      journal={arXiv:2509.21790},\n}",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "神经世界模拟器"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2509.19080",
    "title": "World4RL: Diffusion World Models for Policy Refinement with Reinforcement Learning for Robotic Manipulation",
    "authors": "Zhennan Jiang; Kai Liu; Yuxin Qin; Shuai Tian; Yupeng Zheng; Mingcai Zhou; Chao Yu; Haoran Li; Dongbin Zhao",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2025-09-23",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "world4rl",
    "arxivUrl": "https://arxiv.org/abs/2509.19080",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2509.19080",
    "pdfUrl": "https://arxiv.org/pdf/2509.19080",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@article{world4rl,\n      title={World4RL: Diffusion World Models for Policy Refinement with Reinforcement Learning for Robotic Manipulation}, \n      author={Zhennan Jiang and Kai Liu and Yuxin Qin and Shuai Tian and Yupeng Zheng and Mingcai Zhou and Chao Yu and Haoran Li and Dongbin Zhao},\n      year={2025},\n      journal={arXiv:2509.19080},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "策略后训练与WM-RL"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2509.18428",
    "title": "Latent Action Pretraining Through World Modeling",
    "authors": "Bahey Tharwat; Yara Nasser; Ali Abouzeid; Ian Reid",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2025-09-22",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "lawm",
    "arxivUrl": "https://arxiv.org/abs/2509.18428",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2509.18428",
    "pdfUrl": "https://arxiv.org/pdf/2509.18428",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@article{lawm,\n      title={Latent Action Pretraining Through World Modeling}, \n      author={Bahey Tharwat and Yara Nasser and Ali Abouzeid and Ian Reid},\n      year={2025},\n      journal={arXiv:2509.18428},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "潜动作预训练",
      "训练优化与蒸馏"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2508.15874",
    "title": "Spatial Policy: Guiding Visuomotor Robotic Manipulation with Spatial-Aware Modeling and Reasoning",
    "authors": "Yijun Liu; Yuwei Liu; Yuan Meng; Jieheng Zhang; Yuwei Zhou; Ye Li; Jiacheng Jiang; Kangye Ji; Shijia Ge; Zhi Wang; Wenwu Zhu",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2025-08-21",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "spatial_policy",
    "arxivUrl": "https://arxiv.org/abs/2508.15874",
    "codeUrls": [
      "https://github.com/PlantPotatoOnMoon/SpatialPolicy"
    ],
    "projectUrl": "https://plantpotatoonmoon.github.io/SpatialPolicy/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2508.15874",
    "pdfUrl": "https://arxiv.org/pdf/2508.15874",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@article{spatial_policy,\n      title={Spatial Policy: Guiding Visuomotor Robotic Manipulation with Spatial-Aware Modeling and Reasoning}, \n      author={Yijun Liu and Yuwei Liu and Yuan Meng and Jieheng Zhang and Yuwei Zhou and Ye Li and Jiacheng Jiang and Kangye Ji and Shijia Ge and Zhi Wang and Wenwu Zhu},\n      year={2025},\n      journal={arXiv:2508.15874},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "视觉规划与IDM",
      "三维多视角建模"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2508.14327",
    "title": "MoVieDrive: Urban Scene Synthesis with Multi-Modal Multi-View Video Diffusion Transformer",
    "authors": "Guile Wu; David Huang; Dongfeng Bai; Bingbing Liu",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2025-08-20",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "omviedrive",
    "arxivUrl": "https://arxiv.org/abs/2508.14327",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2508.14327",
    "pdfUrl": "https://arxiv.org/pdf/2508.14327",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@article{omviedrive,\n      title={MoVieDrive: Urban Scene Synthesis with Multi-Modal Multi-View Video Diffusion Transformer}, \n      author={Guile Wu and David Huang and Dongfeng Bai and Bingbing Liu},\n      year={2025},\n      journal={arXiv:2508.14327},\n}",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "神经世界模拟器"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2508.06571",
    "title": "IRL-VLA: Training an Vision-Language-Action Policy via Reward World Model",
    "authors": "Anqing Jiang; Yu Gao; Yiru Wang; Zhigang Sun; Shuo Wang; Yuwen Heng; Hao Sun; Shichen Tang; Lijuan Zhu; Jinhao Chai; Jijun Wang; Zichong Gu; Hao Jiang; Li Sun",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2025-08-07",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "irl_vla",
    "arxivUrl": "https://arxiv.org/abs/2508.06571",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2508.06571",
    "pdfUrl": "https://arxiv.org/pdf/2508.06571",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@article{irl_vla,\n      title={IRL-VLA: Training an Vision-Language-Action Policy via Reward World Model}, \n      author={Anqing Jiang and Yu Gao and Yiru Wang and Zhigang Sun and Shuo Wang and Yuwen Heng and Hao Sun and Shichen Tang and Lijuan Zhu and Jinhao Chai and Jijun Wang and Zichong Gu and Hao Jiang and Li Sun},\n      year={2025},\n      journal={arXiv:2508.06571},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "策略后训练与WM-RL",
      "自动驾驶"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2508.05635",
    "title": "Genie Envisioner: A Unified World Foundation Platform for Robotic Manipulation",
    "authors": "Yue Liao; Pengfei Zhou; Siyuan Huang; Donglin Yang; Shengcong Chen; Yuxin Jiang; Yue Hu; Jingbin Cai; Si Liu; Jianlan Luo; Liliang Chen; Shuicheng Yan; Maoqing Yao; Guanghui Ren",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2025-08-07",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "genie_env",
    "arxivUrl": "https://arxiv.org/abs/2508.05635",
    "codeUrls": [
      "https://github.com/AgibotTech/Genie-Envisioner"
    ],
    "projectUrl": "https://genie-envisioner.github.io/",
    "venue": "ICLR 2026",
    "paperUrl": "https://arxiv.org/abs/2508.05635",
    "pdfUrl": "https://arxiv.org/pdf/2508.05635",
    "doi": null,
    "publicationYear": 2026,
    "bibtex": "@article{genie_env,\n      title={Genie Envisioner: A Unified World Foundation Platform for Robotic Manipulation}, \n      author={Yue Liao and Pengfei Zhou and Siyuan Huang and Donglin Yang and Shengcong Chen and Yuxin Jiang and Yue Hu and Jingbin Cai and Si Liu and Jianlan Luo and Liliang Chen and Shuicheng Yan and Maoqing Yao and Guanghui Ren},\n      year={2025},\n      journal={arXiv:2508.05635},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "泛化与动作对齐",
      "高效推理与实时控制"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2508.00795",
    "title": "Video Generators are Robot Policies",
    "authors": "Junbang Liang; Pavel Tokmakov; Ruoshi Liu; Sruthi Sudhakar; Paarth Shah; Rares Ambrus; Carl Vondrick",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2025-08-01",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "video_policy",
    "arxivUrl": "https://arxiv.org/abs/2508.00795",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2508.00795",
    "pdfUrl": "https://arxiv.org/pdf/2508.00795",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@article{video_policy,\n      title={Video Generators are Robot Policies}, \n      author={Junbang Liang and Pavel Tokmakov and Ruoshi Liu and Sruthi Sudhakar and Paarth Shah and Rares Ambrus and Carl Vondrick},\n      year={2025},\n      journal={arXiv:2508.00795},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "联合视频动作建模",
      "泛化与动作对齐"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "联合预测",
    "quadrant": "Q3 · Dual-system × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2507.12898",
    "title": "Vidar: Embodied Video Diffusion Model for Generalist Manipulation",
    "authors": "Yao Feng; Hengkai Tan; Xinyi Mao; Chendong Xiang; Guodong Liu; Shuhe Huang; Hang Su; Jun Zhu",
    "affiliations": "Dept. of Comp. Sci. and Tech., Institute for AI, BNRist Center, THBI Lab, Tsinghua-Bosch Joint ML Center, Tsinghua University; Shengshu Tech, Beijing, 100084, China",
    "contribution": "Vidar adapts an embodied video generator to a target bimanual robot, then converts predicted imagery into controls using a separately trained masked inverse dynamics model. Low-demonstration real-world results and component ablations support this design; action observability, open-loop execution, and substantial prior training constrain its generality.",
    "abstract": "",
    "submittedDate": "2025-07-17",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv250712898",
    "arxivUrl": "https://arxiv.org/abs/2507.12898",
    "codeUrls": [],
    "projectUrl": null,
    "venue": "arXiv preprint",
    "paperUrl": "https://arxiv.org/abs/2507.12898",
    "pdfUrl": "https://arxiv.org/pdf/2507.12898",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "视觉规划与IDM"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": null
  },
  {
    "id": "2507.05240",
    "title": "StreamVLN: Streaming Vision-and-Language Navigation via SlowFast Context Modeling",
    "authors": "Meng Wei; Chenyang Wan; Xiqian Yu; Tai Wang; Yuqiang Yang; Xiaohan Mao; Chenming Zhu; Wenzhe Cai; Hanqing Wang; Yilun Chen; Xihui Liu; Jiangmiao Pang",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2025-07-07",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "streamvln",
    "arxivUrl": "https://arxiv.org/abs/2507.05240",
    "codeUrls": [
      "https://github.com/InternRobotics/StreamVLN"
    ],
    "projectUrl": "https://streamvln.github.io/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2507.05240",
    "pdfUrl": "https://arxiv.org/pdf/2507.05240",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@article{streamvln,\n      title={StreamVLN: Streaming Vision-and-Language Navigation via SlowFast Context Modeling}, \n      author={Meng Wei and Chenyang Wan and Xiqian Yu and Tai Wang and Yuqiang Yang and Xiaohan Mao and Chenming Zhu and Wenzhe Cai and Hanqing Wang and Yilun Chen and Xihui Liu and Jiangmiao Pang},\n      year={2025},\n      journal={arXiv:2507.05240},\n}",
    "majorCategory": "VLA",
    "subcategories": [
      "自回归VLA",
      "记忆与长时序VLA"
    ],
    "architecture": "One Model",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2507.05169",
    "title": "Critique of World Model",
    "authors": "Eric Xing; Mingkai Deng; Jinyu Hou",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2025-07-07",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "xing2025critiques; xing2025CWM",
    "arxivUrl": "https://arxiv.org/abs/2507.05169",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2507.05169",
    "pdfUrl": "https://arxiv.org/pdf/2507.05169",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@article{xing2025critiques,\n      title={Critique of World Model}, \n      author={Eric Xing and Mingkai Deng and Jinyu Hou},\n      year={2025},\n      journal={arXiv:2507.05169},\n}\n\n@article{xing2025CWM,\n  title={Critique of World Model},\n  author={Xing, Eric and Deng, Mingkai and Hou, Jinyu},\n  journal={arXiv:2507.05169},\n  year={2025}\n}",
    "majorCategory": "奠基性工作",
    "subcategories": [
      "理论与规划",
      "综述与技术资源"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2507.04447",
    "title": "DreamVLA: A Vision-Language-Action Model Dreamed with Comprehensive World Knowledge",
    "authors": "Wenyao Zhang; Hongsi Liu; Zekun Qi; Yunnan Wang; Xinqiang Yu; Jiazhao Zhang; Runpei Dong; Jiawei He; Fan Lu; He Wang; Zhizheng Zhang; Li Yi; Wenjun Zeng; Xin Jin",
    "affiliations": "SJTU; EIT; THU; Galbot; PKU; UIUC; USTC",
    "contribution": "DreamVLA trains a shared backbone to anticipate motion-relevant regions, depth and semantic features, then conditions action diffusion on its latent predictions. Explicit visual decoders disappear at inference. It reports 4.44 completed instructions on CALVIN ABC-D and 76.7% real-robot success under an attempt-limited protocol. Its lesson is selective forecasting, tempered by unresolved mask, configuration and table inconsistencies.",
    "abstract": "",
    "submittedDate": "2025-07-06",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv250704447",
    "arxivUrl": "https://arxiv.org/abs/2507.04447",
    "codeUrls": [],
    "projectUrl": null,
    "venue": "arXiv preprint",
    "paperUrl": "https://arxiv.org/abs/2507.04447",
    "pdfUrl": "https://arxiv.org/pdf/2507.04447",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "视觉规划与IDM",
      "潜空间预测与JEPA"
    ],
    "architecture": null,
    "predictionParadigm": null,
    "quadrant": null,
    "classificationStatus": null
  },
  {
    "id": "2506.23135",
    "title": "RoboScape: Physics-informed Embodied World Model",
    "authors": "Yu Shang; Xin Zhang; Yinzhou Tang; Lei Jin; Chen Gao; Wei Wu; Yong Li",
    "affiliations": "Tsinghua University; Manifold AI",
    "contribution": "RoboScape predicts action-conditioned robotic RGB and depth videos using coupled autoregressive branches. Depth-feature feedback and motion-selected keypoint supervision aim to improve geometry and interaction dynamics. Reported benefits include stronger video metrics, useful synthetic policy-training data, and correlated simulator-based policy evaluation; the ablations expose tradeoffs, and physical robot deployment remains future work.",
    "abstract": "",
    "submittedDate": "2025-06-29",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv250623135",
    "arxivUrl": "https://arxiv.org/abs/2506.23135",
    "codeUrls": [],
    "projectUrl": null,
    "venue": "arXiv preprint",
    "paperUrl": "https://arxiv.org/abs/2506.23135",
    "pdfUrl": "https://arxiv.org/pdf/2506.23135",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "神经世界模拟器",
      "神经策略评测环境"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": null
  },
  {
    "id": "2506.21539",
    "title": "WorldVLA: Towards Autoregressive Action World Model",
    "authors": "Jun Cen; Chaohui Yu; Hangjie Yuan; Yuming Jiang; Siteng Huang; Jiayan Guo; Xin Li; Yibing Song; Hao Luo; Fan Wang; Deli Zhao; Hao Chen",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2025-06-26",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "worldvla",
    "arxivUrl": "https://arxiv.org/abs/2506.21539",
    "codeUrls": [
      "https://github.com/alibaba-damo-academy/RynnVLA-002"
    ],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2506.21539",
    "pdfUrl": "https://arxiv.org/pdf/2506.21539",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@article{worldvla,\n      title={WorldVLA: Towards Autoregressive Action World Model}, \n      author={Jun Cen and Chaohui Yu and Hangjie Yuan and Yuming Jiang and Siteng Huang and Jiayan Guo and Xin Li and Yibing Song and Hao Luo and Fan Wang and Deli Zhao and Hao Chen},\n      year={2025},\n      journal={arXiv:2506.21539},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "联合视频动作建模"
    ],
    "architecture": "One Model",
    "predictionParadigm": "联合预测",
    "quadrant": "Q1 · One Model × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2506.14198",
    "title": "AMPLIFY: Actionless Motion Priors for Robot Learning from Videos",
    "authors": "Jeremy A. Collins; Loránd Cheng; Kunal Aneja; Albert Wilcox; Benjamin Joffe; Animesh Garg",
    "affiliations": "Georgia Tech; Georgia Tech Research Institute",
    "contribution": "AMPLIFY learns a compact vocabulary of visual motion from point tracks, predicts that motion from an image and task instruction, and translates it into robot actions through a separate inverse model. Its strongest evidence concerns scarce target-task action labels, including transfer where target-task videos remain available. Better track prediction and physical task completion are evaluated separately.",
    "abstract": "",
    "submittedDate": "2025-06-17",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv250614198",
    "arxivUrl": "https://arxiv.org/abs/2506.14198",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2506.14198",
    "pdfUrl": "https://arxiv.org/pdf/2506.14198",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "视觉规划与IDM"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": null
  },
  {
    "id": "2506.09985",
    "title": "V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning",
    "authors": "Mido Assran; Adrien Bardes; David Fan; Quentin Garrido; Russell Howes; Mojtaba; Komeili; Matthew Muckley; Ammar Rizvi; Claire Roberts; Koustuv Sinha; Artem Zholus; Sergio Arnaud; Abha Gejji; Ada Martin; Francois Robert Hogan; Daniel Dugas; Piotr Bojanowski; Vasil Khalidov; Patrick Labatut; Francisco Massa; Marc Szafraniec; Kapil Krishnakumar; Yong Li; Xiaodong Ma; Sarath Chandar; Franziska Meier; Yann LeCun; Michael Rabbat; Nicolas Ballas",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2025-06-11",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "assran2025vjepa2selfsupervisedvideo",
    "arxivUrl": "https://arxiv.org/abs/2506.09985",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2506.09985",
    "pdfUrl": "https://arxiv.org/pdf/2506.09985",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@article{assran2025vjepa2selfsupervisedvideo,\n      title={V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning}, \n      author={Mido Assran and Adrien Bardes and David Fan and Quentin Garrido and Russell Howes and Mojtaba and Komeili and Matthew Muckley and Ammar Rizvi and Claire Roberts and Koustuv Sinha and Artem Zholus and Sergio Arnaud and Abha Gejji and Ada Martin and Francois Robert Hogan and Daniel Dugas and Piotr Bojanowski and Vasil Khalidov and Patrick Labatut and Francisco Massa and Marc Szafraniec and Kapil Krishnakumar and Yong Li and Xiaodong Ma and Sarath Chandar and Franziska Meier and Yann LeCun and Michael Rabbat and Nicolas Ballas},\n      year={2025},\n      journal={arXiv:2506.09985},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "潜空间预测与JEPA",
      "泛化与动作对齐"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2506.09113",
    "title": "Seedance 1.0: Exploring the Boundaries of Video Generation Models",
    "authors": "Yu Gao; Haoyuan Guo; Tuyen Hoang; Weilin Huang; Lu Jiang; Fangyuan Kong; Huixia Li; Jiashi Li; Liang Li; Xiaojie Li; Xunsong Li; Yifu Li; Shanchuan Lin; Zhijie Lin; Jiawei Liu; Shu Liu; Xiaonan Nie; Zhiwu Qing; Yuxi Ren; Li Sun; Zhi Tian; Rui Wang; Sen Wang; Guoqiang Wei; Guohong Wu; Jie Wu; Ruiqi Xia; Fei Xiao; Xuefeng Xiao; Jiangqiao Yan; Ceyuan Yang; Jianchao Yang; Runkai Yang; Tao Yang; Yihang Yang; Zilyu Ye; Xuejiao Zeng; Yan Zeng; Heng Zhang; Yang Zhao; Xiaozheng Zheng; Peihao Zhu; Jiaxin Zou; Feilong Zuo",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2025-06-10",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "seedance",
    "arxivUrl": "https://arxiv.org/abs/2506.09113",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2506.09113",
    "pdfUrl": "https://arxiv.org/pdf/2506.09113",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@article{seedance,\n      title={Seedance 1.0: Exploring the Boundaries of Video Generation Models}, \n      author={Yu Gao and Haoyuan Guo and Tuyen Hoang and Weilin Huang and Lu Jiang and Fangyuan Kong and Huixia Li and Jiashi Li and Liang Li and Xiaojie Li and Xunsong Li and Yifu Li and Shanchuan Lin and Zhijie Lin and Jiawei Liu and Shu Liu and Xiaonan Nie and Zhiwu Qing and Yuxi Ren and Li Sun and Zhi Tian and Rui Wang and Sen Wang and Guoqiang Wei and Guohong Wu and Jie Wu and Ruiqi Xia and Fei Xiao and Xuefeng Xiao and Jiangqiao Yan and Ceyuan Yang and Jianchao Yang and Runkai Yang and Tao Yang and Yihang Yang and Zilyu Ye and Xuejiao Zeng and Yan Zeng and Heng Zhang and Yang Zhao and Xiaozheng Zheng and Peihao Zhu and Jiaxin Zou and Feilong Zuo},\n      year={2025},\n      journal={arXiv:2506.09113},\n}",
    "majorCategory": "WAM Components",
    "subcategories": [
      "Video & world prediction backbones",
      "视频生成Backbone"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2506.06199",
    "title": "3DFlowAction: Learning Cross-Embodiment Manipulation from 3D Flow World Model",
    "authors": "Hongyan Zhi; Peihao Chen; Siyuan Zhou; Yubo Dong; Quanxi Wu; Lei Han; Mingkui Tan",
    "affiliations": "South China University of Technology; Tencent Robotics X; Hong Kong University of Science and Technology; Pazhou Laboratory",
    "contribution": "3DFlowAction learns instruction-conditioned object trajectories from human and robot videos, checks a rendered endpoint with GPT-4o, and converts accepted flow into robot poses through grasp selection and optimization. Its four-task physical evaluation reports 70% success, but transfer depends on rigid grasp geometry and follows task-specific human-video fine-tuning.",
    "abstract": "",
    "submittedDate": "2025-06-06",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv250606199",
    "arxivUrl": "https://arxiv.org/abs/2506.06199",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2506.06199",
    "pdfUrl": "https://arxiv.org/pdf/2506.06199",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "视觉规划与IDM",
      "三维多视角建模"
    ],
    "architecture": "待核实",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": null
  },
  {
    "id": "2505.24156",
    "title": "Towards a Generalizable Bimanual Foundation Policy via Flow-based Video Prediction",
    "authors": "Chenyou Fan; Fangzheng Yan; Chenjia Bai; Jiepeng Wang; Chi Zhang; Zhen Wang; Xuelong Li",
    "affiliations": "Institute of Artificial intelligence (TeleAI), China Telecom; Northwestern Polytechnical University; Hong Kong University of Science and Technology",
    "contribution": "CogRobot adapts video diffusion to bimanual control through three learned components: instruction-conditioned flow prediction, flow-conditioned RGB prediction, and a task-specific goal-reaching action policy. Flow offers an intermediate description of motion without requiring action labels for the video models. The strongest physical result is Pull Box success of 0.75 versus 0.05 for DP, but the evidence does not establish a universal action policy or broad unseen-task transfer (e02-framing, e05-video, e06-controller, e09-real, e13-limit).",
    "abstract": "",
    "submittedDate": "2025-05-30",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv250524156",
    "arxivUrl": "https://arxiv.org/abs/2505.24156",
    "codeUrls": [],
    "projectUrl": null,
    "venue": "arXiv preprint",
    "paperUrl": "https://arxiv.org/abs/2505.24156",
    "pdfUrl": "https://arxiv.org/pdf/2505.24156",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "视觉规划与IDM"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": null
  },
  {
    "id": "2505.23171",
    "title": "RoboTransfer: Controllable Geometry-Consistent Video Diffusion for Manipulation Policy Transfer",
    "authors": "Liu Liu; Xiaofeng Wang; Guosheng Zhao; Keyu Li; Wenkang Qin; Jiagang Zhu; Jiaxiong Qiu; Zheng Zhu; Guan Huang; Zhizhong Su",
    "affiliations": "Horizon Robotics; GigaAI; CASIA",
    "contribution": "RoboTransfer augments robot demonstrations by changing their visual appearance while conditioning video diffusion on the demonstrated geometry. Jointly encoded camera views, metric depth, normals and separate background/object references support multi-view synthesis. A separately trained ACT policy benefits from the augmented observations on two physical manipulation tasks. This is evidence for offline data augmentation, with remaining uncertainty about statistical reliability and physical fidelity.",
    "abstract": "",
    "submittedDate": "2025-05-29",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv250523171",
    "arxivUrl": "https://arxiv.org/abs/2505.23171",
    "codeUrls": [],
    "projectUrl": null,
    "venue": "arXiv preprint",
    "paperUrl": "https://arxiv.org/abs/2505.23171",
    "pdfUrl": "https://arxiv.org/pdf/2505.23171",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "",
    "majorCategory": "数据集",
    "subcategories": [
      "合成数据与数据生成"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": null
  },
  {
    "id": "2505.20795",
    "title": "Learning Generalizable Robot Policy with Human Demonstration Video as a Prompt",
    "authors": "Xiang Zhu; Yichen Liu; Hezhong Li; Jianyu Chen",
    "affiliations": "Tsinghua University, China; Shanghai Qi Zhi Institute, China",
    "contribution": "A human demonstration video conditions a frozen video model whose features guide a separate diffusion policy for an Xhand robot. Human hand motions are reconstructed and retargeted into the policy's action space, allowing human demonstrations to supervise action learning. The reported transfer concerns objects and skills absent from robot training demonstrations but present in human training data; it does not establish learning an entirely untrained skill from the inference prompt alone.",
    "abstract": "",
    "submittedDate": "2025-05-27",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv250520795",
    "arxivUrl": "https://arxiv.org/abs/2505.20795",
    "codeUrls": [],
    "projectUrl": null,
    "venue": "ICRA 2026",
    "paperUrl": "https://arxiv.org/abs/2505.20795",
    "pdfUrl": "https://arxiv.org/pdf/2505.20795",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "",
    "majorCategory": "Related resources",
    "subcategories": [
      "Visuomotor policy methods"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": null
  },
  {
    "id": "2505.19017",
    "title": "WorldEval: World Model as Real-World Robot Policies Evaluator",
    "authors": "Yaxuan Li; Yichen Zhu; Junjie Wen; Chaomin Shen; Yi Xu",
    "affiliations": "Midea Group; East China Normal University",
    "contribution": "WorldEval estimates robot-policy rankings by turning internal policy embeddings into generated manipulation videos and judging their outcomes. Policy2Vec conditions a separately adapted WAN 2.1 simulator; Gemini-2.0 supplies success labels. Paired experiments support relative-ranking usefulness on the tested tabletop setup, while imperfect action fidelity, checkpoint-distribution dependence and inconsistent source reporting limit stronger conclusions.",
    "abstract": "",
    "submittedDate": "2025-05-25",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv250519017",
    "arxivUrl": "https://arxiv.org/abs/2505.19017",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2505.19017",
    "pdfUrl": "https://arxiv.org/pdf/2505.19017",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "神经策略评测环境",
      "世界模型评测基准"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": null
  },
  {
    "id": "2505.18650",
    "title": "ProphetDWM: A Driving World Model for Rolling Out Future Actions and Videos",
    "authors": "Xiaodong Wang; Peixi Peng",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2025-05-24",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "prophetdwm",
    "arxivUrl": "https://arxiv.org/abs/2505.18650",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2505.18650",
    "pdfUrl": "https://arxiv.org/pdf/2505.18650",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@article{prophetdwm,\n      title={ProphetDWM: A Driving World Model for Rolling Out Future Actions and Videos}, \n      author={Xiaodong Wang and Peixi Peng},\n      year={2025},\n      journal={arXiv:2505.18650},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "联合视频动作建模",
      "自动驾驶"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "联合预测",
    "quadrant": "Q3 · Dual-system × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2505.15659",
    "title": "FLARE: Robot Learning with Implicit World Modeling",
    "authors": "Ruijie Zheng; Jing Wang; Scott Reed; Johan Bjorck; Yu Fang; Fengyuan Hu; Joel Jang; Kaushil Kundalia; Zongyu Lin; Loic Magne; Avnish Narayan; You Liang Tan; Guanzhi Wang; Qi Wang; Jiannan Xiang; Yinzhen Xu; Seonghyeon Ye; Jan Kautz; Furong Huang; Yuke Zhu; Linxi Fan",
    "affiliations": "NVIDIA; University of Maryland, College Park; Nanyang Technological University; University of Texas, Austin",
    "contribution": "FLARE trains a flow-matching robot policy to match future observation embeddings inside its action-denoising transformer. Compact, action-trained visual-language targets supply an auxiliary learning signal, including from action-free human videos. Simulation and real-robot improvements support this training recipe; they do not establish an explicit planner or calibrated world simulator.",
    "abstract": "",
    "submittedDate": "2025-05-21",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv250515659",
    "arxivUrl": "https://arxiv.org/abs/2505.15659",
    "codeUrls": [],
    "projectUrl": null,
    "venue": "arXiv preprint",
    "paperUrl": "https://arxiv.org/abs/2505.15659",
    "pdfUrl": "https://arxiv.org/pdf/2505.15659",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "潜空间预测与JEPA"
    ],
    "architecture": null,
    "predictionParadigm": null,
    "quadrant": null,
    "classificationStatus": null
  },
  {
    "id": "2505.09723",
    "title": "EnerVerse-AC: Envisioning Embodied Environments with Action Condition",
    "authors": "Yuxin Jiang; Shengcong Chen; Siyuan Huang; Liliang Chen; Pengfei Zhou; Yue Liao; Xindong He; Chiming Liu; Hongsheng Li; Maoqing Yao; Guanghui Ren",
    "affiliations": "AgiBot; SJTU; MMLab-CUHK",
    "contribution": "EVAC turns robot action sequences into future camera observations using a video diffusion model. Spatial pose maps, temporal action differences and camera rays condition the generated environment. A separate policy can interact with that environment for evaluation, or learn from synthetic trajectories. The strongest evidence is a small policy-data augmentation experiment and agreement with real-robot evaluation trends; physical accuracy remains incompletely measured.",
    "abstract": "",
    "submittedDate": "2025-05-14",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv250509723",
    "arxivUrl": "https://arxiv.org/abs/2505.09723",
    "codeUrls": [],
    "projectUrl": null,
    "venue": "arXiv preprint",
    "paperUrl": "https://arxiv.org/abs/2505.09723",
    "pdfUrl": "https://arxiv.org/pdf/2505.09723",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "神经世界模拟器",
      "神经策略评测环境"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": null
  },
  {
    "id": "2505.09694",
    "title": "EWMBench: Evaluating Scene, Motion, and Semantic Quality in Embodied World Models",
    "authors": "Yue Hu; Siyuan Huang; Yue Liao; Shengcong Chen; Pengfei Zhou; Liliang Chen; Maoqing Yao; Guanghui Ren",
    "affiliations": "AgiBot; SJTU; MMLab-CUHK; HIT",
    "contribution": "EWMBench evaluates instruction-conditioned robot videos through scene consistency, end-effector motion, and semantics. Its seven-model comparison favors domain-adapted generators, while controlled trajectory corruptions expose why static-looking plausibility is insufficient. These are offline video-evaluation results, with unresolved reporting inconsistencies, rather than demonstrations of executed robot control.",
    "abstract": "",
    "submittedDate": "2025-05-14",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv250509694",
    "arxivUrl": "https://arxiv.org/abs/2505.09694",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2505.09694",
    "pdfUrl": "https://arxiv.org/pdf/2505.09694",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "世界模型评测基准"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": null
  },
  {
    "id": "2505.06111",
    "title": "UniVLA: Learning to Act Anywhere with Task-centric Latent Actions",
    "authors": "Qingwen Bu; Yanting Yang; Jisong Cai; Shenyuan Gao; Guanghui Ren; Maoqing Yao; Ping Luo; Hongyang Li",
    "affiliations": "The University of Hong Kong; OpenDriveLab; AgiBot",
    "contribution": "UniVLA learns a discrete action vocabulary from language-annotated videos, teaches a vision-language policy to predict that vocabulary, and adapts a visual-conditioned head to physical controls. Its distinguishing idea is to separate task-relevant motion from distracting changes before policy pretraining. Strong manipulation and navigation results support transfer, but deployment still needs action-labeled adaptation; future-observation prediction trains the latent representation rather than serving as the evaluated inference-time planner (e-lam, e-policy, e-decode, e-libero, e-nav, e-limitations).",
    "abstract": "",
    "submittedDate": "2025-05-09",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv250506111",
    "arxivUrl": "https://arxiv.org/abs/2505.06111",
    "codeUrls": [],
    "projectUrl": null,
    "venue": "RSS 2025 (arXiv author comments)",
    "paperUrl": "https://arxiv.org/abs/2505.06111",
    "pdfUrl": "https://arxiv.org/pdf/2505.06111",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "",
    "majorCategory": "VLA",
    "subcategories": [
      "潜动作预训练",
      "自回归VLA"
    ],
    "architecture": null,
    "predictionParadigm": null,
    "quadrant": null,
    "classificationStatus": null
  },
  {
    "id": "2505.04999",
    "title": "CLAM: Continuous Latent Action Models for Robot Learning from Unlabeled Demonstrations",
    "authors": "Anthony Liang; Pavel Czempin; Matthew M. Hong; Yutai Zhou; Jingzhen Wang; Erdem Bıyık; Stephen Tu",
    "affiliations": "Thomas Lord Department of Computer Science, University of Southern California; Ming Hsieh Department of Electrical and Computer Engineering, University of Southern California",
    "contribution": "CLAM learns continuous action codes from robot observation transitions, grounds them with limited action-labeled data, then imitates expert videos in that latent space. Deployment uses a policy and action decoder; future-observation prediction serves training. Its strongest evidence concerns action-label scarcity within one robot embodiment, with several unresolved protocol details.",
    "abstract": "",
    "submittedDate": "2025-05-08",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv250504999",
    "arxivUrl": "https://arxiv.org/abs/2505.04999",
    "codeUrls": [],
    "projectUrl": null,
    "venue": "IEEE/RSJ International Conference on Intelligent Robots and Systems 2026 (arXiv journal reference)",
    "paperUrl": "https://arxiv.org/abs/2505.04999",
    "pdfUrl": "https://arxiv.org/pdf/2505.04999",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "",
    "majorCategory": "Related resources",
    "subcategories": [
      "Latent-action & representation methods"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": null
  },
  {
    "id": "ref-ceb20cc004172847c348",
    "title": "Learned Perceptive Forward Dynamics Model for Safe and Platform-aware Robotic Navigation",
    "authors": "Pascal Roth; Jonas Frey; Cesar Cadena; Marco Hutter",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2025-04-27",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "roth2025learned; RothP-RSS-25",
    "arxivUrl": null,
    "codeUrls": [
      "https://github.com/leggedrobotics/fdm"
    ],
    "projectUrl": "https://leggedrobotics.github.io/fdm.github.io/",
    "venue": "Robotics: Science and Systems (RSS) 2025",
    "paperUrl": "https://www.roboticsproceedings.org/rss21/p001.html",
    "pdfUrl": "http://www.roboticsproceedings.org/rss21/p001.pdf",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@article{roth2025learned,\n  title={Learned perceptive forward dynamics model for safe and platform-aware robotic navigation},\n  author={Roth, Pascal and Frey, Jonas and Cadena, Cesar and Hutter, Marco},\n  journal={arXiv preprint arXiv:2504.19322},\n  year={2025}\n}\n\n@inproceedings{RothP-RSS-25,\n  author    = {Pascal Roth and Jonas Frey and Cesar Cadena and Marco Hutter},\n  title     = {{Learned Perceptive Forward Dynamics Model for Safe and Platform-aware Robotic Navigation}},\n  booktitle = {RSS},\n  year      = {2025},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "导航",
      "三维多视角建模"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2504.16693",
    "title": "PIN-WM: Learning Physics-INformed World Models for Non-Prehensile Manipulation",
    "authors": "Wenxuan Li; Hang Zhao; Zhiyuan Yu; Yu Du; Qin Zou; Ruizhen Hu; Kai Xu",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2025-04-23",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "pin_wm",
    "arxivUrl": "https://arxiv.org/abs/2504.16693",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2504.16693",
    "pdfUrl": "https://arxiv.org/pdf/2504.16693",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@article{pin_wm,\n      title={Pin-wm: Learning physics-informed world models for non-prehensile manipulation}, \n      author={Li, Wenxuan and Zhao, Hang and Yu, Zhiyuan and Du, Yu and Zou, Qin and Hu, Ruizhen and Xu, Kai},\n      year={2025},\n      journal={arXiv:2504.16693},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "三维多视角建模",
      "策略后训练与WM-RL"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2504.02792",
    "title": "Unified World Models: Coupling Video and Action Diffusion for Pretraining on Large Robotic Datasets",
    "authors": "Chuning Zhu; Raymond Yu; Siyuan Feng; Benjamin Burchfiel; Paarth Shah; Abhishek Gupta",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2025-04-03",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "uwm",
    "arxivUrl": "https://arxiv.org/abs/2504.02792",
    "codeUrls": [
      "https://github.com/WEIRDLabUW/unified-world-model"
    ],
    "projectUrl": "https://weirdlabuw.github.io/uwm/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2504.02792",
    "pdfUrl": "https://arxiv.org/pdf/2504.02792",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@article{uwm,\n      title={Unified world models: Coupling video and action diffusion for pretraining on large robotic datasets}, \n      author={Zhu, Chuning and Yu, Raymond and Feng, Siyuan and Burchfiel, Benjamin and Shah, Paarth and Gupta, Abhishek},\n      year={2025},\n      journal={arXiv:2504.02792},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "联合视频动作建模",
      "泛化与动作对齐"
    ],
    "architecture": "One Model",
    "predictionParadigm": "联合预测",
    "quadrant": "Q1 · One Model × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2503.22020",
    "title": "CoT-VLA: Visual Chain-of-Thought Reasoning for Vision-Language-Action Models",
    "authors": "Qingqing Zhao; Yao Lu; Moo Jin Kim; Zipeng Fu; Zhuoyang Zhang; Yecheng Wu; Zhaoshuo Li; Qianli Ma; Song Han; Chelsea Finn; Ankur Handa; Ming-Yu Liu; Donglai Xiang; Gordon Wetzstein; Tsung-Yi Lin",
    "affiliations": "NVIDIA; Stanford University; MIT",
    "contribution": "CoT-VLA makes a 7B multimodal model generate a future subgoal image before predicting a robot action chunk. The shared model learns image prediction from demonstrations and captioned videos, while action supervision requires demonstrations. Experiments support useful goal conditioning and stronger average manipulation performance, with uneven task gains, substantial inference overhead and reporting inconsistencies. Visual chain of thought here means an explicit predicted image used for control; general reasoning is not independently established.",
    "abstract": "",
    "submittedDate": "2025-03-27",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv250322020",
    "arxivUrl": "https://arxiv.org/abs/2503.22020",
    "codeUrls": [],
    "projectUrl": null,
    "venue": "CVPR 2025",
    "paperUrl": "https://arxiv.org/abs/2503.22020",
    "pdfUrl": "https://arxiv.org/pdf/2503.22020",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "视觉规划与IDM"
    ],
    "architecture": "One Model",
    "predictionParadigm": "IDM",
    "quadrant": "Q2 · One Model × IDM",
    "classificationStatus": null
  },
  {
    "id": "2503.20523",
    "title": "GAIA-2: A Controllable Multi-View Generative World Model for Autonomous Driving",
    "authors": "Lloyd Russell; Anthony Hu; Lorenzo Bertoni; George Fedoseev; Jamie Shotton; Elahe Arani; Gianluca Corrado",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2025-03-26",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "gaia_2",
    "arxivUrl": "https://arxiv.org/abs/2503.20523",
    "codeUrls": [],
    "projectUrl": "https://wayve.ai/thinking/gaia-2/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2503.20523",
    "pdfUrl": "https://arxiv.org/pdf/2503.20523",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@article{gaia_2,\n      title={GAIA-2: A Controllable Multi-View Generative World Model for Autonomous Driving}, \n      author={Lloyd Russell and Anthony Hu and Lorenzo Bertoni and George Fedoseev and Jamie Shotton and Elahe Arani and Gianluca Corrado},\n      year={2025},\n      journal={arXiv:2503.20523},\n}",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "神经世界模拟器"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2503.20314",
    "title": "Wan: Open and Advanced Large-Scale Video Generative Models",
    "authors": "Team Wan; Ang Wang; Baole Ai; Bin Wen; Chaojie Mao; Chen-Wei Xie; Di Chen; Feiwu Yu; Haiming Zhao; Jianxiao Yang; Jianyuan Zeng; Jiayu Wang; Jingfeng Zhang; Jingren Zhou; Jinkai Wang; Jixuan Chen; Kai Zhu; Kang Zhao; Keyu Yan; Lianghua Huang; Mengyang Feng; Ningyi Zhang; Pandeng Li; Pingyu Wu; Ruihang Chu; Ruili Feng; Shiwei Zhang; Siyang Sun; Tao Fang; Tianxing Wang; Tianyi Gui; Tingyu Weng; Tong Shen; Wei Lin; Wei Wang; Wei Wang; Wenmeng Zhou; Wente Wang; Wenting Shen; Wenyuan Yu; Xianzhong Shi; Xiaoming Huang; Xin Xu; Yan Kou; Yangyu Lv; Yifei Li; Yijing Liu; Yiming Wang; Yingya Zhang; Yitong Huang; Yong Li; You Wu; Yu Liu; Yulin Pan; Yun Zheng; Yuntao Hong; Yupeng Shi; Yutong Feng; Zeyinzi Jiang; Zhen Han; Zhi-Fan Wu; Ziyu Liu",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2025-03-26",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "wan",
    "arxivUrl": "https://arxiv.org/abs/2503.20314",
    "codeUrls": [
      "https://github.com/Wan-Video/Wan2.1"
    ],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2503.20314",
    "pdfUrl": "https://arxiv.org/pdf/2503.20314",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@article{wan,\n      title={Wan: Open and Advanced Large-Scale Video Generative Models}, \n      author={Team Wan and Ang Wang and Baole Ai and Bin Wen and Chaojie Mao and Chen-Wei Xie and Di Chen and Feiwu Yu and Haiming Zhao and Jianxiao Yang and Jianyuan Zeng and Jiayu Wang and Jingfeng Zhang and Jingren Zhou and Jinkai Wang and Jixuan Chen and Kai Zhu and Kang Zhao and Keyu Yan and Lianghua Huang and Mengyang Feng and Ningyi Zhang and Pandeng Li and Pingyu Wu and Ruihang Chu and Ruili Feng and Shiwei Zhang and Siyang Sun and Tao Fang and Tianxing Wang and Tianyi Gui and Tingyu Weng and Tong Shen and Wei Lin and Wei Wang and Wei Wang and Wenmeng Zhou and Wente Wang and Wenting Shen and Wenyuan Yu and Xianzhong Shi and Xiaoming Huang and Xin Xu and Yan Kou and Yangyu Lv and Yifei Li and Yijing Liu and Yiming Wang and Yingya Zhang and Yitong Huang and Yong Li and You Wu and Yu Liu and Yulin Pan and Yun Zheng and Yuntao Hong and Yupeng Shi and Yutong Feng and Zeyinzi Jiang and Zhen Han and Zhi-Fan Wu and Ziyu Liu},\n      year={2025},\n      journal={arXiv:2503.20314},\n}",
    "majorCategory": "WAM Components",
    "subcategories": [
      "Video & world prediction backbones",
      "视频生成Backbone"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2503.19786",
    "title": "Gemma 3 Technical Report",
    "authors": "Gemma Team; Aishwarya Kamath; Johan Ferret; Shreya Pathak; Nino Vieillard; Ramona Merhej; Sarah Perrin; Tatiana Matejovicova; Alexandre Ramé; Morgane Rivière; Louis Rouillard; Thomas Mesnard; Geoffrey Cideron; Jean-bastien Grill; Sabela Ramos; Edouard Yvinec; Michelle Casbon; Etienne Pot; Ivo Penchev; Gaël Liu; Francesco Visin; Kathleen Kenealy; Lucas Beyer; Xiaohai Zhai; Anton Tsitsulin; Robert Busa-Fekete; Alex Feng; Noveen Sachdeva; Benjamin Coleman; Yi Gao; Basil Mustafa; Iain Barr; Emilio Parisotto; David Tian; Matan Eyal; Colin Cherry; Jan-Thorsten Peter; Danila Sinopalnikov; Surya Bhupatiraju; Rishabh Agarwal; Mehran Kazemi; Dan Malkin; Ravin Kumar; David Vilar; Idan Brusilovsky; Jiaming Luo; Andreas Steiner; Abe Friesen; Abhanshu Sharma; Abheesht Sharma; Adi Mayrav Gilady; Adrian Goedeckemeyer; Alaa Saade; Alex Feng; Alexander Kolesnikov; Alexei Bendebury; Alvin Abdagic; Amit Vadi; András György; André Susano Pinto; Anil Das; Ankur Bapna; Antoine Miech; Antoine Yang; Antonia Paterson; Ashish Shenoy; Ayan Chakrabarti; Bilal Piot; Bo Wu; Bobak Shahriari; Bryce Petrini; Charlie Chen; Charline Le Lan; Christopher A. Choquette-Choo; CJ Carey; Cormac Brick; Daniel Deutsch; Danielle Eisenbud; Dee Cattle; Derek Cheng; Dimitris Paparas; Divyashree Shivakumar Sreepathihalli; Doug Reid; Dustin Tran; Dustin Zelle; Eric Noland; Erwin Huizenga; Eugene Kharitonov; Frederick Liu; Gagik Amirkhanyan; Glenn Cameron; Hadi Hashemi; Hanna Klimczak-Plucińska; Harman Singh; Harsh Mehta; Harshal Tushar Lehri; Hussein Hazimeh; Ian Ballantyne; Idan Szpektor; Ivan Nardini; Jean Pouget-Abadie; Jetha Chan; Joe Stanton; John Wieting; Jonathan Lai; Jordi Orbay; Joseph Fernandez; Josh Newlan; Ju-yeong Ji; Jyotinder Singh; Kat Black; Kathy Yu; Kevin Hui; Kiran Vodrahalli; Klaus Greff; Linhai Qiu; Marcella Valentine; Marina Coelho; Marvin Ritter; Matt Hoffman; Matthew Watson; Mayank Chaturvedi; Michael Moynihan; Min Ma; Nabila Babar; Natasha Noy; Nathan Byrd; Nick Roy; Nikola Momchev; Nilay Chauhan; Noveen Sachdeva; Oskar Bunyan; Pankil Botarda; Paul Caron; Paul Kishan Rubenstein; Phil Culliton; Philipp Schmid; Pier Giuseppe Sessa; Pingmei Xu; Piotr Stanczyk; Pouya Tafti; Rakesh Shivanna; Renjie Wu; Renke Pan; Reza Rokni; Rob Willoughby; Rohith Vallu; Ryan Mullins; Sammy Jerome; Sara Smoot; Sertan Girgin; Shariq Iqbal; Shashir Reddy; Shruti Sheth; Siim Põder; Sijal Bhatnagar; Sindhu Raghuram Panyam; Sivan Eiger; Susan Zhang; Tianqi Liu; Trevor Yacovone; Tyler Liechty; Uday Kalra; Utku Evci; Vedant Misra; Vincent Roseberry; Vlad Feinberg; Vlad Kolesnikov; Woohyun Han; Woosuk Kwon; Xi Chen; Yinlam Chow; Yuvein Zhu; Zichuan Wei; Zoltan Egyed; Victor Cotruta; Minh Giang; Phoebe Kirk; Anand Rao; Kat Black; Nabila Babar; Jessica Lo; Erica Moreira; Luiz Gustavo Martins; Omar Sanseviero; Lucas Gonzalez; Zach Gleicher; Tris Warkentin; Vahab Mirrokni; Evan Senter; Eli Collins; Joelle Barral; Zoubin Ghahramani; Raia Hadsell; Yossi Matias; D. Sculley; Slav Petrov; Noah Fiedel; Noam Shazeer; Oriol Vinyals; Jeff Dean; Demis Hassabis; Koray Kavukcuoglu; Clement Farabet; Elena Buchatskaya; Jean-Baptiste Alayrac; Rohan Anil; Dmitry; Lepikhin; Sebastian Borgeaud; Olivier Bachem; Armand Joulin; Alek Andreev; Cassidy Hardin; Robert Dadashi; Léonard Hussenot",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2025-03-25",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "gemma_2025",
    "arxivUrl": "https://arxiv.org/abs/2503.19786",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2503.19786",
    "pdfUrl": "https://arxiv.org/pdf/2503.19786",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@article{gemma_2025,\n      title={Gemma 3 Technical Report}, \n      author={Gemma Team and Aishwarya Kamath and Johan Ferret and Shreya Pathak and Nino Vieillard and Ramona Merhej and Sarah Perrin and Tatiana Matejovicova and Alexandre Ramé and Morgane Rivière and Louis Rouillard and Thomas Mesnard and Geoffrey Cideron and Jean-bastien Grill and Sabela Ramos and Edouard Yvinec and Michelle Casbon and Etienne Pot and Ivo Penchev and Gaël Liu and Francesco Visin and Kathleen Kenealy and Lucas Beyer and Xiaohai Zhai and Anton Tsitsulin and Robert Busa-Fekete and Alex Feng and Noveen Sachdeva and Benjamin Coleman and Yi Gao and Basil Mustafa and Iain Barr and Emilio Parisotto and David Tian and Matan Eyal and Colin Cherry and Jan-Thorsten Peter and Danila Sinopalnikov and Surya Bhupatiraju and Rishabh Agarwal and Mehran Kazemi and Dan Malkin and Ravin Kumar and David Vilar and Idan Brusilovsky and Jiaming Luo and Andreas Steiner and Abe Friesen and Abhanshu Sharma and Abheesht Sharma and Adi Mayrav Gilady and Adrian Goedeckemeyer and Alaa Saade and Alex Feng and Alexander Kolesnikov and Alexei Bendebury and Alvin Abdagic and Amit Vadi and András György and André Susano Pinto and Anil Das and Ankur Bapna and Antoine Miech and Antoine Yang and Antonia Paterson and Ashish Shenoy and Ayan Chakrabarti and Bilal Piot and Bo Wu and Bobak Shahriari and Bryce Petrini and Charlie Chen and Charline Le Lan and Christopher A. Choquette-Choo and CJ Carey and Cormac Brick and Daniel Deutsch and Danielle Eisenbud and Dee Cattle and Derek Cheng and Dimitris Paparas and Divyashree Shivakumar Sreepathihalli and Doug Reid and Dustin Tran and Dustin Zelle and Eric Noland and Erwin Huizenga and Eugene Kharitonov and Frederick Liu and Gagik Amirkhanyan and Glenn Cameron and Hadi Hashemi and Hanna Klimczak-Plucińska and Harman Singh and Harsh Mehta and Harshal Tushar Lehri and Hussein Hazimeh and Ian Ballantyne and Idan Szpektor and Ivan Nardini and Jean Pouget-Abadie and Jetha Chan and Joe Stanton and John Wieting and Jonathan Lai and Jordi Orbay and Joseph Fernandez and Josh Newlan and Ju-yeong Ji and Jyotinder Singh and Kat Black and Kathy Yu and Kevin Hui and Kiran Vodrahalli and Klaus Greff and Linhai Qiu and Marcella Valentine and Marina Coelho and Marvin Ritter and Matt Hoffman and Matthew Watson and Mayank Chaturvedi and Michael Moynihan and Min Ma and Nabila Babar and Natasha Noy and Nathan Byrd and Nick Roy and Nikola Momchev and Nilay Chauhan and Noveen Sachdeva and Oskar Bunyan and Pankil Botarda and Paul Caron and Paul Kishan Rubenstein and Phil Culliton and Philipp Schmid and Pier Giuseppe Sessa and Pingmei Xu and Piotr Stanczyk and Pouya Tafti and Rakesh Shivanna and Renjie Wu and Renke Pan and Reza Rokni and Rob Willoughby and Rohith Vallu and Ryan Mullins and Sammy Jerome and Sara Smoot and Sertan Girgin and Shariq Iqbal and Shashir Reddy and Shruti Sheth and Siim Põder and Sijal Bhatnagar and Sindhu Raghuram Panyam and Sivan Eiger and Susan Zhang and Tianqi Liu and Trevor Yacovone and Tyler Liechty and Uday Kalra and Utku Evci and Vedant Misra and Vincent Roseberry and Vlad Feinberg and Vlad Kolesnikov and Woohyun Han and Woosuk Kwon and Xi Chen and Yinlam Chow and Yuvein Zhu and Zichuan Wei and Zoltan Egyed and Victor Cotruta and Minh Giang and Phoebe Kirk and Anand Rao and Kat Black and Nabila Babar and Jessica Lo and Erica Moreira and Luiz Gustavo Martins and Omar Sanseviero and Lucas Gonzalez and Zach Gleicher and Tris Warkentin and Vahab Mirrokni and Evan Senter and Eli Collins and Joelle Barral and Zoubin Ghahramani and Raia Hadsell and Yossi Matias and D. Sculley and Slav Petrov and Noah Fiedel and Noam Shazeer and Oriol Vinyals and Jeff Dean and Demis Hassabis and Koray Kavukcuoglu and Clement Farabet and Elena Buchatskaya and Jean-Baptiste Alayrac and Rohan Anil and Dmitry and Lepikhin and Sebastian Borgeaud and Olivier Bachem and Armand Joulin and Alek Andreev and Cassidy Hardin and Robert Dadashi and Léonard Hussenot},\n      year={2025},\n      journal={arXiv:2503.19786},\n}",
    "majorCategory": "WAM Components",
    "subcategories": [
      "Language & vision-language backbones",
      "语言与VLM Backbone"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2503.14734",
    "title": "GR00T N1: An Open Foundation Model for Generalist Humanoid Robots",
    "authors": "NVIDIA; :; Johan Bjorck; Fernando Castañeda; Nikita Cherniadev; Xingye Da; Runyu Ding; Linxi \"Jim\" Fan; Yu Fang; Dieter Fox; Fengyuan Hu; Spencer Huang; Joel Jang; Zhenyu Jiang; Jan Kautz; Kaushil Kundalia; Lawrence Lao; Zhiqi Li; Zongyu Lin; Kevin Lin; Guilin Liu; Edith Llontop; Loic Magne; Ajay Mandlekar; Avnish Narayan; Soroush Nasiriany; Scott Reed; You Liang Tan; Guanzhi Wang; Zu Wang; Jing Wang; Qi Wang; Jiannan Xiang; Yuqi Xie; Yinzhen Xu; Zhenjia Xu; Seonghyeon Ye; Zhiding Yu; Ao Zhang; Hao Zhang; Yizhou Zhao; Ruijie Zheng; Yuke Zhu",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2025-03-18",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "gr00tn1_2025",
    "arxivUrl": "https://arxiv.org/abs/2503.14734",
    "codeUrls": [
      "https://github.com/NVIDIA/Isaac-GR00T"
    ],
    "projectUrl": "https://developer.nvidia.com/isaac/gr00t",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2503.14734",
    "pdfUrl": "https://arxiv.org/pdf/2503.14734",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@article{gr00tn1_2025,\n      title={GR00T N1: An Open Foundation Model for Generalist Humanoid Robots}, \n      author={NVIDIA and : and Johan Bjorck and Fernando Castañeda and Nikita Cherniadev and Xingye Da and Runyu Ding and Linxi \"Jim\" Fan and Yu Fang and Dieter Fox and Fengyuan Hu and Spencer Huang and Joel Jang and Zhenyu Jiang and Jan Kautz and Kaushil Kundalia and Lawrence Lao and Zhiqi Li and Zongyu Lin and Kevin Lin and Guilin Liu and Edith Llontop and Loic Magne and Ajay Mandlekar and Avnish Narayan and Soroush Nasiriany and Scott Reed and You Liang Tan and Guanzhi Wang and Zu Wang and Jing Wang and Qi Wang and Jiannan Xiang and Yuqi Xie and Yinzhen Xu and Zhenjia Xu and Seonghyeon Ye and Zhiding Yu and Ao Zhang and Hao Zhang and Yizhou Zhao and Ruijie Zheng and Yuke Zhu},\n      year={2025},\n      journal={arXiv:2503.14734},\n}",
    "majorCategory": "VLA",
    "subcategories": [
      "扩散与流匹配VLA",
      "分层与双系统VLA"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2503.14492",
    "title": "Cosmos-Transfer1: Conditional World Generation with Adaptive Multimodal Control",
    "authors": "NVIDIA; :; Hassan Abu Alhaija; Jose Alvarez; Maciej Bala; Tiffany Cai; Tianshi Cao; Liz Cha; Joshua Chen; Mike Chen; Francesco Ferroni; Sanja Fidler; Dieter Fox; Yunhao Ge; Jinwei Gu; Ali Hassani; Michael Isaev; Pooya Jannaty; Shiyi Lan; Tobias Lasser; Huan Ling; Ming-Yu Liu; Xian Liu; Yifan Lu; Alice Luo; Qianli Ma; Hanzi Mao; Fabio Ramos; Xuanchi Ren; Tianchang Shen; Xinglong Sun; Shitao Tang; Ting-Chun Wang; Jay Wu; Jiashu Xu; Stella Xu; Kevin Xie; Yuchong Ye; Xiaodong Yang; Xiaohui Zeng; Yu Zeng",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2025-03-18",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "cosmos_transfer",
    "arxivUrl": "https://arxiv.org/abs/2503.14492",
    "codeUrls": [
      "https://github.com/nvidia-cosmos/cosmos-transfer1"
    ],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2503.14492",
    "pdfUrl": "https://arxiv.org/pdf/2503.14492",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@article{cosmos_transfer,\n      title={Cosmos-Transfer1: Conditional World Generation with Adaptive Multimodal Control}, \n      author={NVIDIA and : and Hassan Abu Alhaija and Jose Alvarez and Maciej Bala and Tiffany Cai and Tianshi Cao and Liz Cha and Joshua Chen and Mike Chen and Francesco Ferroni and Sanja Fidler and Dieter Fox and Yunhao Ge and Jinwei Gu and Ali Hassani and Michael Isaev and Pooya Jannaty and Shiyi Lan and Tobias Lasser and Huan Ling and Ming-Yu Liu and Xian Liu and Yifan Lu and Alice Luo and Qianli Ma and Hanzi Mao and Fabio Ramos and Xuanchi Ren and Tianchang Shen and Xinglong Sun and Shitao Tang and Ting-Chun Wang and Jay Wu and Jiashu Xu and Stella Xu and Kevin Xie and Yuchong Ye and Xiaodong Yang and Xiaohui Zeng and Yu Zeng},\n      year={2025},\n      journal={arXiv:2503.14492},\n}",
    "majorCategory": "WAM Components",
    "subcategories": [
      "Video & world prediction backbones",
      "视频生成Backbone"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2503.10370",
    "title": "LUMOS: Language-Conditioned Imitation Learning with World Models",
    "authors": "Iman Nematollahi; Branton DeMoss; Akshay L Chandra; Nick Hawes; Wolfram Burgard; Ingmar Posner",
    "affiliations": "University of Freiburg; University of Oxford; University of Technology Nuremberg",
    "contribution": "LUMOS trains a language-guided manipulation policy by practicing inside a world model learned from offline robot play. Frozen recurrent dynamics support actor-critic learning with expert-latent matching rewards; hindsight plans and language alignment organize behavior. CALVIN and physical tabletop experiments support this combination, with modest gains over adapted HULC and larger component-ablation losses. Transfer means deployment after learning from that environment’s offline data, without online policy fine-tuning.",
    "abstract": "",
    "submittedDate": "2025-03-13",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv250310370",
    "arxivUrl": "https://arxiv.org/abs/2503.10370",
    "codeUrls": [],
    "projectUrl": null,
    "venue": "ICRA 2025 (acceptance stated in official arXiv comments)",
    "paperUrl": "https://arxiv.org/abs/2503.10370",
    "pdfUrl": "https://arxiv.org/pdf/2503.10370",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "策略后训练与WM-RL"
    ],
    "architecture": "待核实",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": null
  },
  {
    "id": "2503.00200",
    "title": "Unified Video Action Model",
    "authors": "Shuang Li; Yihuai Gao; Dorsa Sadigh; Shuran Song",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2025-02-28",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "uva",
    "arxivUrl": "https://arxiv.org/abs/2503.00200",
    "codeUrls": [
      "https://github.com/ShuangLI59/unified_video_action"
    ],
    "projectUrl": "https://unified-video-action-model.github.io/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2503.00200",
    "pdfUrl": "https://arxiv.org/pdf/2503.00200",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@article{uva,\n      title={Unified video action model}, \n      author={Li, Shuang and Gao, Yihuai and Sadigh, Dorsa and Song, Shuran},\n      year={2025},\n      journal={arXiv:2503.00200},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "联合视频动作建模",
      "高效推理与实时控制"
    ],
    "architecture": "One Model",
    "predictionParadigm": "联合预测",
    "quadrant": "Q1 · One Model × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2502.19645",
    "title": "Fine-Tuning Vision-Language-Action Models: Optimizing Speed and Success",
    "authors": "Moo Jin Kim; Chelsea Finn; Percy Liang",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2025-02-27",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "openvla_oft",
    "arxivUrl": "https://arxiv.org/abs/2502.19645",
    "codeUrls": [
      "https://github.com/moojink/openvla-oft"
    ],
    "projectUrl": "https://openvla-oft.github.io/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2502.19645",
    "pdfUrl": "https://arxiv.org/pdf/2502.19645",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@article{openvla_oft,\n      title={Fine-tuning vision-language-action models: Optimizing speed and success}, \n      author={Kim, Moo Jin and Finn, Chelsea and Liang, Percy},\n      year={2025},\n      journal={arXiv:2502.19645},\n}",
    "majorCategory": "VLA",
    "subcategories": [
      "VLA后训练与数据增强"
    ],
    "architecture": "One Model",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2502.15672",
    "title": "VaViM and VaVAM: Autonomous Driving through Video Generative Modeling",
    "authors": "Florent Bartoccioni; Elias Ramzi; Victor Besnier; Shashanka Venkataramanan; Tuan-Hung Vu; Yihong Xu; Loick Chambon; Spyros Gidaris; Serkan Odabas; David Hurych; Renaud Marlet; Alexandre Boulch; Mickael Chen; Éloi Zablocki; Andrei Bursuc; Eduardo Valle; Matthieu Cord",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2025-02-21",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "vavim_vavam",
    "arxivUrl": "https://arxiv.org/abs/2502.15672",
    "codeUrls": [
      "https://github.com/valeoai/VideoActionModel"
    ],
    "projectUrl": "https://valeoai.github.io/vavim-vavam/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2502.15672",
    "pdfUrl": "https://arxiv.org/pdf/2502.15672",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@article{vavim_vavam,\n      title={VaViM and VaVAM: Autonomous Driving through Video Generative Modeling}, \n      author={Florent Bartoccioni and Elias Ramzi and Victor Besnier and Shashanka Venkataramanan and Tuan-Hung Vu and Yihong Xu and Loick Chambon and Spyros Gidaris and Serkan Odabas and David Hurych and Renaud Marlet and Alexandre Boulch and Mickael Chen and Éloi Zablocki and Andrei Bursuc and Eduardo Valle and Matthieu Cord},\n      year={2025},\n      journal={arXiv:2502.15672},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "自动驾驶",
      "高效推理与实时控制"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2502.10498",
    "title": "The Role of World Models in Shaping Autonomous Driving: A Comprehensive Survey",
    "authors": "Sifan Tu; Xin Zhou; Dingkang Liang; Xingyu Jiang; Yumeng Zhang; Xiaofan Li; Xiang Bai",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2025-02-14",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "comprehensive_dwm_survey",
    "arxivUrl": "https://arxiv.org/abs/2502.10498",
    "codeUrls": [
      "https://github.com/LMD0311/Awesome-World-Model"
    ],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2502.10498",
    "pdfUrl": "https://arxiv.org/pdf/2502.10498",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@article{comprehensive_dwm_survey,\n      title={The Role of World Models in Shaping Autonomous Driving: A Comprehensive Survey}, \n      author={Sifan Tu and Xin Zhou and Dingkang Liang and Xingyu Jiang and Yumeng Zhang and Xiaofan Li and Xiang Bai},\n      year={2025},\n      journal={arXiv:2502.10498},\n}",
    "majorCategory": "Related resources",
    "subcategories": [
      "综述与技术资源"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2502.01784",
    "title": "VILP: Imitation Learning with Latent Video Planning",
    "authors": "Zhengtong Xu, Qiang Qiu, Yu She",
    "affiliations": "School of Industrial Engineering, Purdue University, West Lafayette, USA; Elmore Family School of Electrical and Computer Engineering, Purdue University, West Lafayette, USA",
    "contribution": "VILP learns observation-conditioned future videos in a compressed latent space, decodes them, and maps adjacent frames to actions through a separate low-level policy. This makes short-horizon video planning practical on the tested tasks and lets task videos supply information beyond scarce action labels. Simulation gains depend on data and evaluation protocol; real-robot evidence comprises 15 trials per method.",
    "abstract": "",
    "submittedDate": "2025-02-03",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv250201784",
    "arxivUrl": "https://arxiv.org/abs/2502.01784",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2502.01784",
    "pdfUrl": "https://arxiv.org/pdf/2502.01784",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "视觉规划与IDM"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": null
  },
  {
    "id": "2501.09781",
    "title": "VideoWorld: Exploring Knowledge Learning from Unlabeled Videos",
    "authors": "Zhongwei Ren; Yunchao Wei; Xun Guo; Yao Zhao; Bingyi Kang; Jiashi Feng; Xiaojie Jin",
    "affiliations": "Beijing Jiaotong University; ByteDance Seed",
    "contribution": "VideoWorld predicts compact multi-step dynamics codes alongside video frames, then converts predictions into task operations. On rendered 9×9 Go and simulated robot tasks, this representation substantially improves over video-only prediction. Expert-curated training data, language conditioning and a separately supervised robot action decoder qualify the broader claim of learning solely from unlabeled videos.",
    "abstract": "",
    "submittedDate": "2025-01-16",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv250109781",
    "arxivUrl": "https://arxiv.org/abs/2501.09781",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2501.09781",
    "pdfUrl": "https://arxiv.org/pdf/2501.09781",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "视觉规划与IDM",
      "潜空间预测与JEPA"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": null
  },
  {
    "id": "2501.09747",
    "title": "FAST: Efficient Action Tokenization for Vision-Language-Action Models",
    "authors": "Karl Pertsch; Kyle Stachowicz; Brian Ichter; Danny Driess; Suraj Nair; Quan Vuong; Oier Mees; Chelsea Finn; Sergey Levine",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2025-01-16",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "fast",
    "arxivUrl": "https://arxiv.org/abs/2501.09747",
    "codeUrls": [
      "https://huggingface.co/physical-intelligence/fast"
    ],
    "projectUrl": "https://www.pi.website/research/fast",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2501.09747",
    "pdfUrl": "https://arxiv.org/pdf/2501.09747",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@article{fast,\n      title={Fast: Efficient action tokenization for vision-language-action models}, \n      author={Pertsch, Karl and Stachowicz, Kyle and Ichter, Brian and Driess, Danny and Nair, Suraj and Vuong, Quan and Mees, Oier and Finn, Chelsea and Levine, Sergey},\n      year={2025},\n      journal={arXiv:2501.09747},\n}",
    "majorCategory": "WAM Components",
    "subcategories": [
      "Action representations & policies",
      "动作策略基础"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2501.06605",
    "title": "RoboHorizon: An LLM-Assisted Multi-View World Model for Long-Horizon Robotic Manipulation",
    "authors": "Zixuan Chen; Jing Huo; Yangtao Chen; Yang Gao",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2025-01-11",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "robohorizon",
    "arxivUrl": "https://arxiv.org/abs/2501.06605",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2501.06605",
    "pdfUrl": "https://arxiv.org/pdf/2501.06605",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@article{robohorizon,\n      title={Robohorizon: An llm-assisted multi-view world model for long-horizon robotic manipulation}, \n      author={Chen, Zixuan and Huo, Jing and Chen, Yangtao and Gao, Yang},\n      year={2025},\n      journal={arXiv:2501.06605},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "策略后训练与WM-RL",
      "记忆与长时序",
      "三维多视角建模"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2501.01895",
    "title": "EnerVerse: Envisioning Embodied Future Space for Robotics Manipulation",
    "authors": "Siyuan Huang; Liliang Chen; Pengfei Zhou; Shengcong Chen; Yue Liao; Zhengkai Jiang; Yue Hu; Peng Gao; Hongsheng Li; Maoqing Yao; Guanghui Ren",
    "affiliations": "SJTU; AgiBot; Shanghai AI Lab; CUHK MMLab; LV-NUS Lab",
    "contribution": "EnerVerse learns instruction-conditioned, multi-view future-video representations, then conditions a separate action diffusion head on their backbone features. Sparse history supports long tasks; an offline 4D Gaussian Splatting loop refines training videos. Its strongest benchmark average uses depth-rendered auxiliary views, while physical block placement exposes an instruction-following weakness.",
    "abstract": "",
    "submittedDate": "2025-01-03",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv250101895",
    "arxivUrl": "https://arxiv.org/abs/2501.01895",
    "codeUrls": [],
    "projectUrl": null,
    "venue": "NeurIPS 2025",
    "paperUrl": "https://arxiv.org/abs/2501.01895",
    "pdfUrl": "https://arxiv.org/pdf/2501.01895",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "三维多视角建模",
      "记忆与长时序"
    ],
    "architecture": null,
    "predictionParadigm": null,
    "quadrant": null,
    "classificationStatus": null
  },
  {
    "id": "ref-fd9fc6adf5b4623bf71d",
    "title": "Learning Robot Manipulation from Audio World Models",
    "authors": "Fan Zhang; Michael Gienger",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "audio_wm",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": "https://hri-eu.github.io/flow-matching-policy/",
    "venue": null,
    "paperUrl": "https://openreview.net/forum?id=z3DcbKdjom",
    "pdfUrl": "https://arxiv.org/pdf/2512.08405",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@inproceedings{\naudio_wm,\ntitle={Learning Robot Manipulation from Audio World Models},\nauthor={Fan Zhang and Michael Gienger},\nbooktitle={ICRA Workshop},\nyear={2026},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "多模态触觉音频",
      "视觉规划与IDM"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-eb71134f4ab2c037bb71",
    "title": "DriveDreamer-2: LLM-Enhanced World Models for Diverse Driving Video Generation",
    "authors": "Guosheng Zhao; Xiaofeng Wang; Zheng Zhu; Xinze Chen; Guan Huang; Xiaoyi Bao; Xingang Wang",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "drivedreamer_2",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "AAAI 2025",
    "paperUrl": "https://ojs.aaai.org/index.php/AAAI/article/view/33130",
    "pdfUrl": "https://ojs.aaai.org/index.php/AAAI/article/download/33130/35285",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@inproceedings{drivedreamer_2,\n  author       = {Guosheng Zhao and\n                  Xiaofeng Wang and\n                  Zheng Zhu and\n                  Xinze Chen and\n                  Guan Huang and\n                  Xiaoyi Bao and\n                  Xingang Wang},\n  title        = {DriveDreamer-2: LLM-Enhanced World Models for Diverse Driving Video\n                  Generation},\n  booktitle    = {AAAI},\n  pages        = {10412--10420},\n  year         = {2025},\n\n}",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "神经世界模拟器"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-ea14fce8991bbd0da18e",
    "title": "A review of learning-based dynamics models for robotic manipulation",
    "authors": "Bo Ai; Stephen Tian; Haochen Shi; Yixuan Wang; Tobias Pfaff; Cheston Tan; Henrik I Christensen; Hao Su; Jiajun Wu; Yunzhu Li",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "ai2025review",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": "https://www.hichristensen.com/publication/scirob-25-survey/",
    "venue": "Science Robotics 10(106), 2025",
    "paperUrl": "https://doi.org/10.1126/scirobotics.adt1497",
    "pdfUrl": "https://albertboai.com/assets/pdf/2025_scirobotics.adt1497.pdf",
    "doi": "https://doi.org/10.1126/scirobotics.adt1497",
    "publicationYear": 2025,
    "bibtex": "@article{ai2025review,\n  title={A review of learning-based dynamics models for robotic manipulation},\n  author={Ai, Bo and Tian, Stephen and Shi, Haochen and Wang, Yixuan and Pfaff, Tobias and Tan, Cheston and Christensen, Henrik I and Su, Hao and Wu, Jiajun and Li, Yunzhu},\n  journal={Science Robotics},\n  volume={10},\n  number={106},\n  year={2025},\n}",
    "majorCategory": "Related resources",
    "subcategories": [
      "综述与技术资源"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-dbeba1302a9344e1a92a",
    "title": "DreamGen: Unlocking Generalization in Robot Learning through Video World Models",
    "authors": "Joel Jang; Seonghyeon Ye; Zongyu Lin; Jiannan Xiang; Johan Bjorck; Yu Fang; Fengyuan Hu; Spencer Huang; Kaushil Kundalia; Yen-Chen Lin; Lo\\\"ıc Magne; Ajay Mandlekar; Avnish Narayan; You Liang Tan; Guanzhi Wang; Jing Wang; Qi Wang; Yinzhen Xu; Xiaohui Zeng; Kaiyuan Zheng; Ruijie Zheng; Ming-Yu Liu; Luke Zettlemoyer; Dieter Fox; Jan Kautz; Scott Reed; Yuke Zhu; Linxi Fan",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "dreamgen",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "CoRL 2025",
    "paperUrl": "https://proceedings.mlr.press/v305/jang25a.html",
    "pdfUrl": "https://arxiv.org/pdf/2505.12705",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@inproceedings{\ndreamgen,\ntitle={DreamGen: Unlocking Generalization in Robot Learning through Video World Models},\nauthor={Joel Jang and Seonghyeon Ye and Zongyu Lin and Jiannan Xiang and Johan Bjorck and Yu Fang and Fengyuan Hu and Spencer Huang and Kaushil Kundalia and Yen-Chen Lin and Lo{\\\"\\i}c Magne and Ajay Mandlekar and Avnish Narayan and You Liang Tan and Guanzhi Wang and Jing Wang and Qi Wang and Yinzhen Xu and Xiaohui Zeng and Kaiyuan Zheng and Ruijie Zheng and Ming-Yu Liu and Luke Zettlemoyer and Dieter Fox and Jan Kautz and Scott Reed and Yuke Zhu and Linxi Fan},\nbooktitle={9th Annual Conference on Robot Learning},\nyear={2025},\n}",
    "majorCategory": "数据集",
    "subcategories": [
      "合成数据与数据生成",
      "伪动作标注",
      "机器人示范与操作数据"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-d8b16d47d9d8bb62207f",
    "title": "Latent Action Pretraining from Videos",
    "authors": "Seonghyeon Ye; Joel Jang; Byeongguk Jeon; Sejune Joo; Jianwei Yang; Baolin Peng; Ajay Mandlekar; Reuben Tan; Yu-Wei Chao; Bill Yuchen Lin; Lars Liden; Kimin Lee; Jianfeng Gao; Luke Zettlemoyer; Dieter Fox; Minjoon Seo",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "lapa",
    "arxivUrl": null,
    "codeUrls": [
      "https://github.com/LatentActionPretraining/LAPA"
    ],
    "projectUrl": "https://latentactionpretraining.github.io/",
    "venue": "ICLR 2025",
    "paperUrl": "https://iclr.cc/virtual/2025/poster/29409",
    "pdfUrl": "https://arxiv.org/pdf/2410.11758",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@inproceedings{lapa,\n  author       = {Seonghyeon Ye and\n                  Joel Jang and\n                  Byeongguk Jeon and\n                  Se June Joo and\n                  Jianwei Yang and\n                  Baolin Peng and\n                  Ajay Mandlekar and\n                  Reuben Tan and\n                  Yu{-}Wei Chao and\n                  Bill Yuchen Lin and\n                  Lars Liden and\n                  Kimin Lee and\n                  Jianfeng Gao and\n                  Luke Zettlemoyer and\n                  Dieter Fox and\n                  Minjoon Seo},\n  title        = {Latent Action Pretraining from Videos},\n  booktitle    = {ICLR},\n  year         = {2025},\n}",
    "majorCategory": "WAM Components",
    "subcategories": [
      "Action representations & policies",
      "潜动作预训练"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-d54645624df87f5b0163",
    "title": "Precise and Dexterous Robotic Manipulation via Human-in-the-Loop Reinforcement Learning",
    "authors": "Jianlan Luo; Charles Xu; Jeffrey Wu; Sergey Levine",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "hil_serl",
    "arxivUrl": null,
    "codeUrls": [
      "https://github.com/rail-berkeley/hil-serl"
    ],
    "projectUrl": "https://hil-serl.github.io/",
    "venue": "Science Robotics",
    "paperUrl": "https://www.science.org/doi/10.1126/scirobotics.ads5033",
    "pdfUrl": "https://arxiv.org/pdf/2410.21845",
    "doi": "https://doi.org/10.1126/scirobotics.ads5033",
    "publicationYear": 2025,
    "bibtex": "@article{hil_serl,\n  author       = {Jianlan Luo and\n                  Charles Xu and\n                  Jeffrey Wu and\n                  Sergey Levine},\n  title        = {Precise and dexterous robotic manipulation via human-in-the-loop reinforcement\n                  learning},\n  journal      = {Sci. Robot.},\n  volume       = {10},\n  number       = {105},\n  year         = {2025},\n}",
    "majorCategory": "奠基性工作",
    "subcategories": [
      "动作策略基础",
      "训练优化与蒸馏"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-ce6916f81594a9c4c73f",
    "title": "Mean Flows for One-step Generative Modeling",
    "authors": "Zhengyang Geng; Mingyang Deng; Xingjian Bai; Zico Kolter; Kaiming He",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "mean_flow",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "NeurIPS 2025",
    "paperUrl": "https://papers.neurips.cc/paper_files/paper/2025/hash/6d13e085b79d454da5910e4ca82a3d9d-Abstract-Conference.html",
    "pdfUrl": "https://proceedings.neurips.cc/paper_files/paper/2025/file/6d13e085b79d454da5910e4ca82a3d9d-Paper-Conference.pdf",
    "doi": "https://doi.org/10.52202/085713-2534",
    "publicationYear": 2025,
    "bibtex": "@inproceedings{mean_flow,\n author = {Geng, Zhengyang and Deng, Mingyang and Bai, Xingjian and Kolter, Zico and He, Kaiming},\n booktitle = {NeurIPS},\n pages = {75460--75482},\n title = {Mean Flows for One-step Generative Modeling},\n volume = {38},\n year = {2025}\n}",
    "majorCategory": "奠基性工作",
    "subcategories": [
      "扩散与流匹配基础"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-cb61c489d1333f433fc4",
    "title": "Helix: A Vision-Language-Action Model for Generalist Humanoid Control",
    "authors": "Figure AI",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "figure_helix",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": "https://www.figure.ai/news/helix",
    "venue": null,
    "paperUrl": "https://www.figure.ai/news/helix",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@misc{figure_helix,\n    title = {Helix: A Vision-Language-Action Model for Generalist Humanoid Control},\n    author = {Figure AI},\n    year = {2025},\n    howpublished = {\\url{https://www.figure.ai/news/helix}},\n}",
    "majorCategory": "VLA",
    "subcategories": [
      "分层与双系统VLA"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "综述明确"
  },
  {
    "id": "ref-ca883d875395dd7ff120",
    "title": "AdaWorld: Learning Adaptable World Models with Latent Actions",
    "authors": "Shenyuan Gao; Siyuan Zhou; Yilun Du; Jun Zhang; Chuang Gan",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "adaworld",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "ICML 2025",
    "paperUrl": "https://proceedings.mlr.press/v267/gao25u.html",
    "pdfUrl": "https://raw.githubusercontent.com/mlresearch/v267/main/assets/gao25u/gao25u.pdf",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@inproceedings{adaworld,\n  author       = {Shenyuan Gao and\n                  Siyuan Zhou and\n                  Yilun Du and\n                  Jun Zhang and\n                  Chuang Gan},\n  title        = {AdaWorld: Learning Adaptable World Models with Latent Actions},\n  booktitle    = {ICML},\n  year         = {2025},\n \n}",
    "majorCategory": "奠基性工作",
    "subcategories": [
      "潜动作预训练",
      "神经世界模拟器"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-c3cbdc6867eb310bf60b",
    "title": "Mixture-of-Transformers: A Sparse and Scalable Architecture for Multi-Modal Foundation Models",
    "authors": "Weixin Liang; Lili Yu; Liang Luo; Srinivasan Iyer; Ning Dong; Chunting Zhou; Gargi Ghosh; Mike Lewis; Wen-tau Yih; Luke Zettlemoyer; Xi Victoria Lin",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "mot",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "Transactions on Machine Learning Research",
    "paperUrl": "https://openreview.net/forum?id=Nu6N69i8SB",
    "pdfUrl": "https://openreview.net/pdf/98c54be9c1d612834460a51ddcc23866dc7314c9.pdf",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@article{mot,\n  author       = {Weixin Liang and\n                  Lili Yu and\n                  Liang Luo and\n                  Srini Iyer and\n                  Ning Dong and\n                  Chunting Zhou and\n                  Gargi Ghosh and\n                  Mike Lewis and\n                  Wen{-}tau Yih and\n                  Luke Zettlemoyer and\n                  Xi Victoria Lin},\n  title        = {Mixture-of-Transformers: {A} Sparse and Scalable Architecture for\n                  Multi-Modal Foundation Models},\n  journal      = {TMLR},\n  volume       = {2025},\n  year         = {2025},\n  \n}",
    "majorCategory": "WAM Components",
    "subcategories": [
      "Language & vision-language backbones",
      "语言与VLM Backbone"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-ae98866e6b155b0f253c",
    "title": "FAST-LIVO2: Fast, Direct LiDAR-Inertial-Visual Odometry",
    "authors": "Chunran Zheng; Wei Xu; Zuhao Zou; Tong Hua; Chongjian Yuan; Dongjiao He; Bingyang Zhou; Zheng Liu; Jiarong Lin; Fangcheng Zhu; Yunfan Ren; Rong Wang; Fanle Meng; Fu Zhang",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "Zheng2024FASTLIVO2Fast",
    "arxivUrl": null,
    "codeUrls": [
      "https://github.com/hku-mars/FAST-LIVO2"
    ],
    "projectUrl": null,
    "venue": "IEEE Transactions on Robotics",
    "paperUrl": "https://ieeexplore.ieee.org/document/10757429/",
    "pdfUrl": "https://arxiv.org/pdf/2408.14035",
    "doi": "https://doi.org/10.1109/tro.2024.3502198",
    "publicationYear": 2025,
    "bibtex": "@article{Zheng2024FASTLIVO2Fast,\n  author       = {Chunran Zheng and\n                  Wei Xu and\n                  Zuhao Zou and\n                  Tong Hua and\n                  Chongjian Yuan and\n                  Dongjiao He and\n                  Bingyang Zhou and\n                  Zheng Liu and\n                  Jiarong Lin and\n                  Fangcheng Zhu and\n                  Yunfan Ren and\n                  Rong Wang and\n                  Fanle Meng and\n                  Fu Zhang},\n  title        = {{FAST-LIVO2:} Fast, Direct LiDAR-Inertial-Visual Odometry},\n  journal      = {TRO},\n  volume       = {41},\n  pages        = {326--346},\n  year         = {2025},\n\n}",
    "majorCategory": "Related resources",
    "subcategories": [
      "三维表示与状态估计"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-a8d6b1d31a7e72a424da",
    "title": "RLVR-World: Training World Models with Reinforcement Learning",
    "authors": "Jialong Wu; Shaofeng Yin; Ningya Feng; Mingsheng Long",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "rlvr_world",
    "arxivUrl": null,
    "codeUrls": [
      "https://github.com/thuml/RLVR-World"
    ],
    "projectUrl": "https://thuml.github.io/RLVR-World/",
    "venue": "NeurIPS 2025",
    "paperUrl": "https://proceedings.neurips.cc/paper_files/paper/2025/hash/b63a24a1832bd14fa945c71f535c0095-Abstract-Conference.html",
    "pdfUrl": "https://proceedings.neurips.cc/paper_files/paper/2025/file/b63a24a1832bd14fa945c71f535c0095-Paper-Conference.pdf",
    "doi": "https://doi.org/10.52202/085713-4177",
    "publicationYear": 2025,
    "bibtex": "@inproceedings{\nrlvr_world,\ntitle={{RLVR}-World: Training World Models with Reinforcement Learning},\nauthor={Jialong Wu and Shaofeng Yin and Ningya Feng and Mingsheng Long},\nbooktitle={NeurIPS},\nyear={2025},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "神经世界模拟器",
      "训练优化与蒸馏"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-a4279cce45a54f73d7ba",
    "title": "ParticleFormer: A 3D Point Cloud World Model for Multi-Object, Multi-Material Robotic Manipulation",
    "authors": "Suning Huang; Qianzhong Chen; Xiaohan Zhang; Jiankai Sun; Mac Schwager",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "particleformer",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": "https://suninghuang19.github.io/particleformer_page/",
    "venue": "CoRL 2025",
    "paperUrl": "https://proceedings.mlr.press/v305/huang25c.html",
    "pdfUrl": "https://proceedings.mlr.press/v305/huang25c/huang25c.pdf",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@inproceedings{particleformer,\ntitle={ParticleFormer: A 3D Point Cloud World Model for Multi-Object, Multi-Material Robotic Manipulation},\nauthor={Suning Huang and Qianzhong Chen and Xiaohan Zhang and Jiankai Sun and Mac Schwager},\nbooktitle={CoRL},\nyear={2025},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "三维多视角建模"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-a351afa5504418f3e26f",
    "title": "π₀.₅: a Vision-Language-Action Model with Open-World Generalization",
    "authors": "Kevin Black; Noah Brown; James Darpinian; Karan Dhabalia; Danny Driess; Adnan Esmail; Michael Robert Equi; Chelsea Finn; Niccolo Fusai; Manuel Y. Galliker; Dibya Ghosh; Lachy Groom; Karol Hausman; brian ichter; Szymon Jakubczak; Tim Jones; Liyiming Ke; Devin LeBlanc; Sergey Levine; Adrian Li-Bell; Mohith Mothukuri; Suraj Nair; Karl Pertsch; Allen Z. Ren; Lucy Xiaoyang Shi; Laura Smith; Jost Tobias Springenberg; Kyle Stachowicz; James Tanner; Quan Vuong; Homer Walke; Anna Walling; Haohuan Wang; Lili Yu; Ury Zhilinsky",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "pi0_5",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "CoRL 2025",
    "paperUrl": "https://proceedings.mlr.press/v305/black25a.html",
    "pdfUrl": "https://raw.githubusercontent.com/mlresearch/v305/main/assets/black25a/black25a.pdf",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@InProceedings{pi0_5,\n  title = \t {$\\pi_{0.5}$: a Vision-Language-Action Model with Open-World Generalization},\n  author =       {Black, Kevin and Brown, Noah and Darpinian, James and Dhabalia, Karan and Driess, Danny and Esmail, Adnan and Equi, Michael Robert and Finn, Chelsea and Fusai, Niccolo and Galliker, Manuel Y. and Ghosh, Dibya and Groom, Lachy and Hausman, Karol and ichter, brian and Jakubczak, Szymon and Jones, Tim and Ke, Liyiming and LeBlanc, Devin and Levine, Sergey and Li-Bell, Adrian and Mothukuri, Mohith and Nair, Suraj and Pertsch, Karl and Ren, Allen Z. and Shi, Lucy Xiaoyang and Smith, Laura and Springenberg, Jost Tobias and Stachowicz, Kyle and Tanner, James and Vuong, Quan and Walke, Homer and Walling, Anna and Wang, Haohuan and Yu, Lili and Zhilinsky, Ury},\n  booktitle = \t {CoRL},\n  pages = \t {17--40},\n  year = \t {2025},\n  volume = \t {305},\n}",
    "majorCategory": "VLA",
    "subcategories": [
      "扩散与流匹配VLA",
      "分层与双系统VLA"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "综述明确"
  },
  {
    "id": "ref-9112f4a198ea02dfd27c",
    "title": "DrivingGPT: Unifying Driving World Modeling and Planning with Multi-modal Autoregressive Transformers",
    "authors": "Yuntao Chen; Yuqi Wang; Zhaoxiang Zhang",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "drivinggpt",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": "https://rogerchern.github.io/DrivingGPT/",
    "venue": "ICCV 2025",
    "paperUrl": "https://openaccess.thecvf.com/content/ICCV2025/html/Chen_DrivingGPT_Unifying_Driving_World_Modeling_and_Planning_with_Multi-modal_Autoregressive_ICCV_2025_paper.html",
    "pdfUrl": "https://arxiv.org/pdf/2412.18607",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@inproceedings{drivinggpt,\n  author       = {Yuntao Chen and\n                  Yuqi Wang and\n                  Zhaoxiang Zhang},\n  title        = {DrivingGPT: Unifying Driving World Modeling and Planning with Multi-Modal\n                  Autoregressive Transformers},\n  booktitle    = {ICCV},\n  pages        = {26890--26900},\n  year         = {2025},\n\n\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "自动驾驶",
      "联合视频动作建模"
    ],
    "architecture": "One Model",
    "predictionParadigm": "联合预测",
    "quadrant": "Q1 · One Model × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-8fa0ebc722d8d35eaf75",
    "title": "Mastering diverse control tasks through world models",
    "authors": "Danijar Hafner; Jurgis Pasukonis; Jimmy Ba; Timothy Lillicrap",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "dreamerv3",
    "arxivUrl": null,
    "codeUrls": [
      "https://github.com/danijar/dreamerv3"
    ],
    "projectUrl": "https://danijar.com/project/dreamerv3/",
    "venue": "Nature 2025",
    "paperUrl": "https://www.nature.com/articles/s41586-025-08744-2",
    "pdfUrl": "https://www.nature.com/articles/s41586-025-08744-2.pdf",
    "doi": "https://doi.org/10.1038/s41586-025-08744-2",
    "publicationYear": 2025,
    "bibtex": "@article{dreamerv3,\n  title     = {Mastering Diverse Control Tasks through World Models},\n  author    = {Hafner, Danijar and Pasukonis, Jurgis and Ba, Jimmy and Lillicrap, Timothy},\n  journal   = {Nature},\n  volume    = {640},\n  pages     = {647--653},\n  year      = {2025},\n}",
    "majorCategory": "奠基性工作",
    "subcategories": [
      "经典WM与模型式RL"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-7fd707e94672f79511d9",
    "title": "World Model-based Perception for Visual Legged Locomotion",
    "authors": "Hang Lai; Jiahang Cao; Jiafeng Xu; Hongtao Wu; Yunfeng Lin; Tao Kong; Yong Yu; Weinan Zhang",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "lai2025world",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "2025 IEEE International Conference on Robotics and Automation (ICRA)",
    "paperUrl": "https://ieeexplore.ieee.org/document/11128762",
    "pdfUrl": "https://arxiv.org/pdf/2409.16784",
    "doi": "https://doi.org/10.1109/icra55743.2025.11128762",
    "publicationYear": 2025,
    "bibtex": "@inproceedings{lai2025world,\n  title={World model-based perception for visual legged locomotion},\n  author={Lai, Hang and Cao, Jiahang and Xu, Jiafeng and Wu, Hongtao and Lin, Yunfeng and Kong, Tao and Yu, Yong and Zhang, Weinan},\n  booktitle={ICRA},\n  pages={11531--11537},\n  year={2025},\n  organization={IEEE}\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "导航",
      "潜空间预测与JEPA"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-797a519828bd34873731",
    "title": "DINO-WM: World Models on Pre-trained Visual Features enable Zero-shot Planning",
    "authors": "Gaoyue Zhou; Hengkai Pan; Yann LeCun; Lerrel Pinto",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "dino_wm",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "ICML 2025",
    "paperUrl": "https://proceedings.mlr.press/v267/zhou25t.html",
    "pdfUrl": "https://arxiv.org/pdf/2411.04983",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@inproceedings{dino_wm,\n  author       = {Gaoyue Zhou and\n                  Hengkai Pan and\n                  Yann LeCun and\n                  Lerrel Pinto},\n  title        = {{DINO-WM:} World Models on Pre-trained Visual Features enable Zero-shot\n                  Planning},\n  booktitle    = {ICML},\n  year         = {2025},\n}",
    "majorCategory": "奠基性工作",
    "subcategories": [
      "经典WM与模型式RL",
      "视觉编码器与表征"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-76e692b7c04cea08a682",
    "title": "Evaluating Real-World Robot Manipulation Policies in Simulation",
    "authors": "Xuanlin Li; Kyle Hsu; Jiayuan Gu; Oier Mees; Karl Pertsch; Homer Rich Walke; Chuyuan Fu; Ishikaa Lunawat; Isabel Sieh; Sean Kirmani; Sergey Levine; Jiajun Wu; Chelsea Finn; Hao Su; Quan Vuong; Ted Xiao",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "simplerenv",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "CoRL 2024（论文集出版于2025年）",
    "paperUrl": "https://proceedings.mlr.press/v270/li25c.html",
    "pdfUrl": "https://raw.githubusercontent.com/mlresearch/v270/main/assets/li25c/li25c.pdf",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@inproceedings{simplerenv,\n  author       = {Xuanlin Li and\n                  Kyle Hsu and\n                  Jiayuan Gu and\n                  Oier Mees and\n                  Karl Pertsch and\n                  Homer Rich Walke and\n                  Chuyuan Fu and\n                  Ishikaa Lunawat and\n                  Isabel Sieh and\n                  Sean Kirmani and\n                  Sergey Levine and\n                  Jiajun Wu and\n                  Chelsea Finn and\n                  Hao Su and\n                  Quan Vuong and\n                  Ted Xiao},\n  title        = {Evaluating Real-World Robot Manipulation Policies in Simulation},\n  booktitle    = {CoRL},\n  pages        = {3705--3728},\n  year         = {2024},\n  \n}",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "机器人操作基准",
      "物理仿真",
      "仿真到真实评测"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "综述明确"
  },
  {
    "id": "ref-65dfbbe844d739cf5492",
    "title": "Epona: Autoregressive Diffusion World Model for Autonomous Driving",
    "authors": "Kaiwen Zhang; Zhenyu Tang; Xiaotao Hu; Xingang Pan; Xiaoyang Guo; Yuan Liu; Jingwei Huang; Li Yuan; Qian Zhang; Xiao-Xiao Long; Xun Cao; Wei Yin",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "epona",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "ICCV 2025",
    "paperUrl": "https://openaccess.thecvf.com/content/ICCV2025/html/Zhang_Epona_Autoregressive_Diffusion_World_Model_for_Autonomous_Driving_ICCV_2025_paper.html",
    "pdfUrl": "https://arxiv.org/pdf/2506.24113",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@inproceedings{epona,\n  author       = {Kaiwen Zhang and\n                  Zhenyu Tang and\n                  Xiaotao Hu and\n                  Xingang Pan and\n                  Xiaoyang Guo and\n                  Yuan Liu and\n                  Jingwei Huang and\n                  Li Yuan and\n                  Qian Zhang and\n                  Xiao{-}Xiao Long and\n                  Xun Cao and\n                  Wei Yin},\n  title        = {Epona: Autoregressive Diffusion World Model for Autonomous Driving},\n  booktitle    = {ICCV},\n  pages        = {27220--27230},\n  year         = {2025},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "自动驾驶",
      "联合视频动作建模",
      "记忆与长时序"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-649842d13a08a63bd6ec",
    "title": "Robotic World Model: A Neural Network Simulator for Robust Policy Optimization in Robotics",
    "authors": "Chenhao Li; Andreas Krause; Marco Hutter",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "robotic_wm; li2025robotic",
    "arxivUrl": null,
    "codeUrls": [
      "https://github.com/leggedrobotics/robotic_world_model"
    ],
    "projectUrl": null,
    "venue": "CoRL 2025 Workshop on Learning to Simulate Robot Worlds",
    "paperUrl": "https://openreview.net/forum?id=u76d3gBWCX",
    "pdfUrl": "https://arxiv.org/pdf/2501.10100",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@inproceedings{\nrobotic_wm,\ntitle={Robotic World Model: A Neural Network Simulator for Robust Policy Optimization in Robotics},\nauthor={Chenhao Li and Andreas Krause and Marco Hutter},\nbooktitle={NeurIPS Workshop},\nyear={2025},\n}\n\n@inproceedings{\nli2025robotic,\ntitle={Robotic World Model: A Neural Network Simulator for Robust Policy Optimization in Robotics},\nauthor={Chenhao Li and Andreas Krause and Marco Hutter},\nbooktitle={CoRL Workshop},\nyear={2025},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "策略后训练与WM-RL",
      "记忆与长时序"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-646db90b75f1827a347d",
    "title": "EgoBridge: Domain Adaptation for Generalizable Imitation from Egocentric Human Data",
    "authors": "Ryan Punamiya; Dhruv Patel; Patcharapong Aphiwetsa; Pranav Kuppili; Lawrence Zhu; Simar Kareer; Judy Hoffman; Danfei Xu",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "Punamiya2025EgoBridgeDomain",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "NeurIPS 2025",
    "paperUrl": "https://neurips.cc/virtual/2025/poster/119049",
    "pdfUrl": "https://papers.neurips.cc/paper_files/paper/2025/file/434756a76c44e220d5f32d6e0b488c0c-Paper-Conference.pdf",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@inproceedings{Punamiya2025EgoBridgeDomain,\n author = {Punamiya, Ryan and Patel, Dhruv and Aphiwetsa, Patcharapong and Kuppili, Pranav and Zhu, Lawrence and Kareer, Simar and Hoffman, Judy and Xu, Danfei},\n booktitle = {NeurIPS},\n pages = {47092--47122},\n title = {EgoBridge: Domain Adaptation for Generalizable Imitation from Egocentric Human Data},\n volume = {38},\n year = {2025}\n}",
    "majorCategory": "Related resources",
    "subcategories": [
      "动作策略基础",
      "训练优化与蒸馏"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-60b7c3c992c8a8c4561d",
    "title": "World4Drive: End-to-End Autonomous Driving via Intention-aware Physical Latent World Model",
    "authors": "Yupeng Zheng; Pengxuan Yang; Zebin Xing; Qichao Zhang; Yuhang Zheng; Yinfeng Gao; Pengfei Li; Teng Zhang; Zhongpu Xia; Peng Jia; XianPeng Lang; Dongbin Zhao",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "world4drive",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "ICCV 2025",
    "paperUrl": "https://openaccess.thecvf.com/content/ICCV2025/html/Zheng_World4Drive_End-to-End_Autonomous_Driving_via_Intention-aware_Physical_Latent_World_Model_ICCV_2025_paper.html",
    "pdfUrl": "https://arxiv.org/pdf/2507.00603",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@inproceedings{world4drive,\n  author       = {Yupeng Zheng and\n                  Pengxuan Yang and\n                  Zebin Xing and\n                  Qichao Zhang and\n                  Yuhang Zheng and\n                  Yinfeng Gao and\n                  Pengfei Li and\n                  Teng Zhang and\n                  Zhongpu Xia and\n                  Peng Jia and\n                  Xianpeng Lang and\n                  Dongbin Zhao},\n  title        = {World4Drive: End-to-End Autonomous Driving via Intention-Aware Physical\n                  Latent World Model},\n  booktitle    = {ICCV},\n  pages        = {28632--28642},\n  year         = {2025},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "自动驾驶",
      "潜空间预测与JEPA"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-5d03f3be7549eb8453b0",
    "title": "A Survey on Vision-Language-Action Models for Autonomous Driving",
    "authors": "Sicong Jiang; Zilin Huang; Kangan Qian; Ziang Luo; Tianze Zhu; Yang Zhong; Yihong Tang; Menglin Kong; Yunlong Wang; Siwen Jiao; Hao Ye; Zihao Sheng; Xin Zhao; Tuopu Wen; Zheng Fu; Sikai Chen; Kun Jiang; Diange Yang; Seongjin Choi; Lijun Sun",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "vla_survey_ad",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "ICCV 2025 Workshops (WDFM-AD)",
    "paperUrl": "https://openaccess.thecvf.com/content/ICCV2025W/WDFM-AD/html/Jiang_A_Survey_on_Vision-Language-Action_Models_for_Autonomous_Driving_ICCVW_2025_paper.html",
    "pdfUrl": "https://openaccess.thecvf.com/content/ICCV2025W/WDFM-AD/papers/Jiang_A_Survey_on_Vision-Language-Action_Models_for_Autonomous_Driving_ICCVW_2025_paper.pdf",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@inproceedings{vla_survey_ad,\n  author       = {Sicong Jiang and\n                  Menglin Kong and\n                  Yihong Tang and\n                  Lijun Sun and\n                  Kangan Qian and\n                  Ziang Luo and\n                  Tianze Zhu and\n                  Yunlong Wang and\n                  Xin Zhao and\n                  Tuopu Wen and\n                  Zheng Fu and\n                  Yang Zhong and\n                  Siwen Jiao and\n                  Hao Ye and\n                  Zilin Huang and\n                  Zihao Sheng and\n                  Sikai Chen and\n                  Seongjin Choi and\n                  Kun Jiang and\n                  Diange Yang},\n  title        = {A Survey on Vision-Language-Action Models for Autonomous Driving},\n  booktitle    = {ICCVW},\n  pages        = {4583--4595},\n  year         = {2025},\n  \n}",
    "majorCategory": "Related resources",
    "subcategories": [
      "综述与技术资源"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-5b11681e66cd6e44cdaf",
    "title": "AutoVLA: A Vision-Language-Action Model for End-to-End Autonomous Driving with Adaptive Reasoning and Reinforcement Fine-Tuning",
    "authors": "Zewei Zhou; Tianhui Cai; Seth Zhao; Yun Zhang; Zhiyu Huang; Bolei Zhou; Jiaqi Ma",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "autovla",
    "arxivUrl": null,
    "codeUrls": [
      "https://github.com/ucla-mobility/AutoVLA"
    ],
    "projectUrl": null,
    "venue": "NeurIPS 2025",
    "paperUrl": "https://proceedings.neurips.cc/paper_files/paper/2025/hash/2843fccca5bedd369a4764848b9bd546-Abstract-Conference.html",
    "pdfUrl": "https://arxiv.org/pdf/2506.13757",
    "doi": "https://doi.org/10.52202/085713-0942",
    "publicationYear": 2025,
    "bibtex": "@article{autovla,\n  title={Autovla: A vision-language-action model for end-to-end autonomous driving with adaptive reasoning and reinforcement fine-tuning},\n  author={Zhou, Zewei and Cai, Tianhui and Zhao, Seth and Zhang, Yun and Huang, Zhiyu and Zhou, Bolei and Ma, Jiaqi},\n  journal={NeurIPS},\n  volume={38},\n  pages={27920--27956},\n  year={2026}\n}",
    "majorCategory": "VLA",
    "subcategories": [
      "自回归VLA",
      "VLA后训练与数据增强"
    ],
    "architecture": "One Model",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-5826406f56d71ab6b5ea",
    "title": "Predictive Inverse Dynamics Models are Scalable Learners for Robotic Manipulation",
    "authors": "Yang Tian; Sizhe Yang; Jia Zeng; Ping Wang; Dahua Lin; Hao Dong; Jiangmiao Pang",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "seer",
    "arxivUrl": null,
    "codeUrls": [
      "https://github.com/InternRobotics/Seer"
    ],
    "projectUrl": "https://nimolty.github.io/Seer/",
    "venue": "ICLR 2025",
    "paperUrl": "https://iclr.cc/virtual/2025/poster/28455",
    "pdfUrl": "https://arxiv.org/pdf/2412.15109",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@inproceedings{seer,\n  author       = {Yang Tian and\n                  Sizhe Yang and\n                  Jia Zeng and\n                  Ping Wang and\n                  Dahua Lin and\n                  Hao Dong and\n                  Jiangmiao Pang},\n  title        = {Predictive Inverse Dynamics Models are Scalable Learners for Robotic\n                  Manipulation},\n  booktitle    = {ICLR},\n  year         = {2025},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "视觉规划与IDM"
    ],
    "architecture": "One Model",
    "predictionParadigm": "IDM",
    "quadrant": "Q2 · One Model × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-502ccb43b26687d765d7",
    "title": "π₀: A Vision-Language-Action Flow Model for General Robot Control",
    "authors": "Kevin Black; Noah Brown; Danny Driess; Adnan Esmail; Michael Robert Equi; Chelsea Finn; Niccolo Fusai; Lachy Groom; Karol Hausman; Brian Ichter; Szymon Jakubczak; Tim Jones; Liyiming Ke; Sergey Levine; Adrian Li-Bell; Mohith Mothukuri; Suraj Nair; Karl Pertsch; Lucy Xiaoyang Shi; Laura Smith; James Tanner; Quan Vuong; Anna Walling; Haohuan Wang; Ury Zhilinsky",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "pi0",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "RSS 2025",
    "paperUrl": "https://www.roboticsproceedings.org/rss21/p010.html",
    "pdfUrl": "https://www.roboticsproceedings.org/rss21/p010.pdf",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@inproceedings{pi0,\n  title = {{$pi\\_0 $: A Vision-Language-Action Flow Model for General Robot Control}},\n  author = {Kevin Black and Noah Brown and Danny Driess and Adnan Esmail and Michael Robert Equi and Chelsea Finn and Niccolo Fusai and Lachy Groom and Karol Hausman and Brian Ichter and Szymon Jakubczak and Tim Jones and Liyiming Ke and Sergey Levine and Adrian Li-Bell and Mohith Mothukuri and Suraj Nair and Karl Pertsch and Lucy Xiaoyang Shi and Laura Smith and James Tanner and Quan Vuong and Anna Walling and Haohuan Wang and Ury Zhilinsky},\n  booktitle = {RSS},\n  year = {2025}\n}",
    "majorCategory": "VLA",
    "subcategories": [
      "扩散与流匹配VLA",
      "分层与双系统VLA"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-4aab09abc95513cc8eb4",
    "title": "RoboTwin: Dual-Arm Robot Benchmark with Generative Digital Twins",
    "authors": "Yao Mu; Tianxing Chen; Zanxin Chen; Shijia Peng; Zhiqian Lan; Zeyu Gao; Zhixuan Liang; Qiaojun Yu; Yude Zou; Mingkun Xu; Lunkai Lin; Zhiqiang Xie; Mingyu Ding; Ping Luo",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "robotwin",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": "https://robotwin-benchmark.github.io",
    "venue": "CVPR 2025",
    "paperUrl": "https://openaccess.thecvf.com/content/CVPR2025/papers/Mu_RoboTwin_Dual-Arm_Robot_Benchmark_with_Generative_Digital_Twins_CVPR_2025_paper.pdf",
    "pdfUrl": "https://openaccess.thecvf.com/content/CVPR2025/papers/Mu_RoboTwin_Dual-Arm_Robot_Benchmark_with_Generative_Digital_Twins_CVPR_2025_paper.pdf",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@inproceedings{robotwin,\n  author       = {Yao Mu and\n                  Tianxing Chen and\n                  Zanxin Chen and\n                  Shijia Peng and\n                  Zhiqian Lan and\n                  Zeyu Gao and\n                  Zhixuan Liang and\n                  Qiaojun Yu and\n                  Yude Zou and\n                  Mingkun Xu and\n                  Lunkai Lin and\n                  Zhiqiang Xie and\n                  Mingyu Ding and\n                  Ping Luo},\n  title        = {RoboTwin: Dual-Arm Robot Benchmark with Generative Digital Twins},\n  booktitle    = {CVPR},\n  pages        = {27649--27660},\n  year         = {2025},\n}",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "机器人操作基准",
      "物理仿真",
      "合成数据与数据生成"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-46bc35f7a73fd1b4e17a",
    "title": "FMB: A functional manipulation benchmark for generalizable robotic learning",
    "authors": "Jianlan Luo; Charles Xu; Fangchen Liu; Liam Tan; Zipeng Lin; Jeffrey Wu; Pieter Abbeel; Sergey Levine",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "fmb",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "The International Journal of Robotics Research",
    "paperUrl": "https://journals.sagepub.com/doi/10.1177/02783649241276017",
    "pdfUrl": "https://arxiv.org/pdf/2401.08553",
    "doi": "https://doi.org/10.1177/02783649241276017",
    "publicationYear": 2025,
    "bibtex": "@article{fmb,\n  author       = {Jianlan Luo and\n                  Charles Xu and\n                  Fangchen Liu and\n                  Liam Tan and\n                  Zipeng Lin and\n                  Jeffrey Wu and\n                  Pieter Abbeel and\n                  Sergey Levine},\n  title        = {{FMB:} {A} functional manipulation benchmark for generalizable robotic\n                  learning},\n  journal      = {IJRR},\n  volume       = {44},\n  number       = {4},\n  pages        = {592--606},\n  year         = {2025},\n\n}",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "机器人操作基准",
      "泛化评测"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-46709651cfd0edc4993a",
    "title": "CogVideoX: Text-to-Video Diffusion Models with An Expert Transformer",
    "authors": "Zhuoyi Yang; Jiayan Teng; Wendi Zheng; Ming Ding; Shiyu Huang; Jiazheng Xu; Yuanming Yang; Wenyi Hong; Xiaohan Zhang; Guanyu Feng; Da Yin; Yuxuan Zhang; Weihan Wang; Yean Cheng; Bin Xu; Xiaotao Gu; Yuxiao Dong; Jie Tang",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "yang2025cogvideox",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "ICLR 2025",
    "paperUrl": "https://openreview.net/forum?id=LQzN6TRFg9",
    "pdfUrl": "https://arxiv.org/pdf/2408.06072",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@inproceedings{yang2025cogvideox,\n  author       = {Zhuoyi Yang and\n                  Jiayan Teng and\n                  Wendi Zheng and\n                  Ming Ding and\n                  Shiyu Huang and\n                  Jiazheng Xu and\n                  Yuanming Yang and\n                  Wenyi Hong and\n                  Xiaohan Zhang and\n                  Guanyu Feng and\n                  Da Yin and\n                  Yuxuan Zhang and\n                  Weihan Wang and\n                  Yean Cheng and\n                  Bin Xu and\n                  Xiaotao Gu and\n                  Yuxiao Dong and\n                  Jie Tang},\n  title        = {CogVideoX: Text-to-Video Diffusion Models with An Expert Transformer},\n  booktitle    = {ICLR},\n  year         = {2025},\n}",
    "majorCategory": "WAM Components",
    "subcategories": [
      "Video & world prediction backbones",
      "视频生成Backbone"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-4482db2c9a502c2e3bae",
    "title": "A0: An Affordance-Aware Hierarchical Model for General Robotic Manipulation",
    "authors": "Rongtao Xu; Jian Zhang; Minghao Guo; Youpeng Wen; Haoting Yang; Min Lin; Jianzheng Huang; Zhe Li; Kaidong Zhang; Liqiong Wang; Yuxuan Kuang; Meng Cao; Feng Zheng; Xiaodan Liang",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "xu2025a0",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "ICCV 2025",
    "paperUrl": "https://openaccess.thecvf.com/content/ICCV2025/html/Xu_A0_An_Affordance-Aware_Hierarchical_Model_for_General_Robotic_Manipulation_ICCV_2025_paper.html",
    "pdfUrl": "https://openaccess.thecvf.com/content/ICCV2025/papers/Xu_A0_An_Affordance-Aware_Hierarchical_Model_for_General_Robotic_Manipulation_ICCV_2025_paper.pdf",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@inproceedings{xu2025a0,\n  title={A0: An affordance-aware hierarchical model for general robotic manipulation},\n  author={Xu, Rongtao and Zhang, Jian and Guo, Minghao and Wen, Youpeng and Yang, Haoting and Lin, Min and Huang, Jianzheng and Li, Zhe and Zhang, Kaidong and Wang, Liqiong and others},\n  booktitle={ICCV},\n  pages={13491--13501},\n  year={2025}\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "三维多视角建模",
      "视觉规划与IDM"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-393a36f38d60db8631f4",
    "title": "Diffusion policy: Visuomotor policy learning via action diffusion",
    "authors": "Cheng Chi; Zhenjia Xu; Siyuan Feng; Eric Cousineau; Yilun Du; Benjamin Burchfiel; Russ Tedrake; Shuran Song",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "diffusion_policy",
    "arxivUrl": null,
    "codeUrls": [
      "https://github.com/real-stanford/diffusion_policy"
    ],
    "projectUrl": "https://diffusion-policy.cs.columbia.edu/",
    "venue": "International Journal of Robotics Research 2025",
    "paperUrl": "https://journals.sagepub.com/doi/10.1177/02783649241273668",
    "pdfUrl": "https://diffusion-policy.cs.columbia.edu/diffusion_policy_ijrr.pdf",
    "doi": "https://doi.org/10.1177/02783649241273668",
    "publicationYear": 2025,
    "bibtex": "@article{diffusion_policy,\n  author       = {Cheng Chi and\n                  Zhenjia Xu and\n                  Siyuan Feng and\n                  Eric Cousineau and\n                  Yilun Du and\n                  Benjamin Burchfiel and\n                  Russ Tedrake and\n                  Shuran Song},\n  title        = {Diffusion policy: Visuomotor policy learning via action diffusion},\n  journal      = {IJRR},\n  volume       = {44},\n  number       = {10-11},\n  pages        = {1684--1704},\n  year         = {2025},\n}",
    "majorCategory": "奠基性工作",
    "subcategories": [
      "动作策略基础",
      "扩散与流匹配基础"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-361c60313b366d2109f0",
    "title": "Pseudo-Simulation for Autonomous Driving",
    "authors": "Wei Cao; Marcel Hallgarten; Tianyu Li; Daniel Dauner; Xunjiang Gu; Caojun Wang; Yakov Miron; Marco Aiello; Hongyang Li; Igor Gilitschenski; Boris Ivanovic; Marco Pavone; Andreas Geiger; Kashyap Chitta",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "navsimv2",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "CoRL 2025",
    "paperUrl": "https://proceedings.mlr.press/v305/cao25a.html",
    "pdfUrl": "https://proceedings.mlr.press/v305/cao25a/cao25a.pdf",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@inproceedings{\nnavsimv2,\ntitle={Pseudo-Simulation for Autonomous Driving},\nauthor={Wei Cao and Marcel Hallgarten and Tianyu Li and Daniel Dauner and Xunjiang Gu and Caojun Wang and Yakov Miron and Marco Aiello and Hongyang Li and Igor Gilitschenski and Boris Ivanovic and Marco Pavone and Andreas Geiger and Kashyap Chitta},\nbooktitle={CoRL},\nyear={2025},\n}",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "自动驾驶基准",
      "评测协议与诊断",
      "驾驶仿真"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-2b3f47a14556997eb476",
    "title": "Video Prediction Policy: A Generalist Robot Policy with Predictive Visual Representations",
    "authors": "Yucheng Hu; Yanjiang Guo; Pengchao Wang; Xiaoyu Chen; Yen-Jen Wang; Jianke Zhang; Koushil Sreenath; Chaochao Lu; Jianyu Chen",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "vpp",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "ICML 2025",
    "paperUrl": "https://proceedings.mlr.press/v267/hu25g.html",
    "pdfUrl": "https://raw.githubusercontent.com/mlresearch/v267/main/assets/hu25g/hu25g.pdf",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@inproceedings{vpp,\n  author       = {Yucheng Hu and\n                  Yanjiang Guo and\n                  Pengchao Wang and\n                  Xiaoyu Chen and\n                  Yen{-}Jen Wang and\n                  Jianke Zhang and\n                  Koushil Sreenath and\n                  Chaochao Lu and\n                  Jianyu Chen},\n  title        = {Video Prediction Policy: {A} Generalist Robot Policy with Predictive\n                  Visual Representations},\n  booktitle    = {ICML},\n  year         = {2025},\n  \n}",
    "majorCategory": "WAM",
    "subcategories": [
      "视觉规划与IDM",
      "高效推理与实时控制"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-283b7da95c4145cf56d0",
    "title": "DyWA: Dynamics-adaptive World Action Model for Generalizable Non-prehensile Manipulation",
    "authors": "Jiangran Lyu; Ziming Li; Xuesong Shi; Chaoyi Xu; Yizhou Wang; He Wang",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "dywa",
    "arxivUrl": null,
    "codeUrls": [
      "https://github.com/jiangranlv/DyWA"
    ],
    "projectUrl": "https://pku-epic.github.io/DyWA/",
    "venue": "ICCV 2025",
    "paperUrl": "https://openaccess.thecvf.com/content/ICCV2025/html/Lyu_DyWA_Dynamics-adaptive_World_Action_Model_for_Generalizable_Non-prehensile_Manipulation_ICCV_2025_paper.html",
    "pdfUrl": "https://openaccess.thecvf.com/content/ICCV2025/papers/Lyu_DyWA_Dynamics-adaptive_World_Action_Model_for_Generalizable_Non-prehensile_Manipulation_ICCV_2025_paper.pdf",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@inproceedings{dywa,\n  author       = {Jiangran Lyu and\n                  Ziming Li and\n                  Xuesong Shi and\n                  Chaoyi Xu and\n                  Yizhou Wang and\n                  He Wang},\n  title        = {DyWA: Dynamics-Adaptive World Action Model for Generalizable Non-Prehensile\n                  Manipulation},\n  booktitle    = {ICCV},\n  pages        = {11058--11068},\n  year         = {2025},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "三维多视角建模",
      "联合视频动作建模",
      "泛化与动作对齐"
    ],
    "architecture": "One Model",
    "predictionParadigm": "联合预测",
    "quadrant": "Q1 · One Model × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-2624494a0f9cbb4a61d2",
    "title": "Navigation World Models",
    "authors": "Amir Bar; Gaoyue Zhou; Danny Tran; Trevor Darrell; Yann LeCun",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "nwm",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "CVPR 2025",
    "paperUrl": "https://openaccess.thecvf.com/content/CVPR2025/html/Bar_Navigation_World_Models_CVPR_2025_paper.html",
    "pdfUrl": "https://openaccess.thecvf.com/content/CVPR2025/papers/Bar_Navigation_World_Models_CVPR_2025_paper.pdf",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@inproceedings{nwm,\n  author    = {Bar, Amir and Zhou, Gaoyue and Tran, Dung and Darrell, Trevor and LeCun, Yann},\n  title     = {Navigation World Models},\n  booktitle = {CVPR},\n  pages     = {15791--15801},\n  year      = {2025},\n  month     = jun,\n  publisher = {IEEE}\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "导航"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-219f4f2dd6b959414001",
    "title": "Cross-Embodiment Dexterous Manipulation through World Model Learning",
    "authors": "Zihao He; Bo Ai; Yulin Liu; Weikang Wan; Henrik I Christensen; Hao Su",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "xembody_wm",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "CoRL 2025 Workshop on Learning to Simulate Robot Worlds",
    "paperUrl": "https://openreview.net/forum?id=1P9LOCLE2G",
    "pdfUrl": "https://openreview.net/pdf?id=1P9LOCLE2G",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@inproceedings{xembody_wm,\ntitle={Cross-Embodiment Dexterous Manipulation through World Model Learning},\nauthor={Zihao He and Bo Ai and Yulin Liu and Weikang Wan and Henrik I Christensen and Hao Su},\nbooktitle={CoRL Workshop},\nyear={2025},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "三维多视角建模",
      "泛化与动作对齐"
    ],
    "architecture": "待核实",
    "predictionParadigm": "待核实",
    "quadrant": "待核实",
    "classificationStatus": "部分待核实"
  },
  {
    "id": "ref-19cb56396363219612a9",
    "title": "GWM: Towards Scalable Gaussian World Models for Robotic Manipulation",
    "authors": "Guanxing Lu; Baoxiong Jia; Puhao Li; Yixin Chen; Ziwei Wang; Yansong Tang; Siyuan Huang",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "gwm",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "ICCV 2025",
    "paperUrl": "https://openaccess.thecvf.com/content/ICCV2025/html/Lu_GWM_Towards_Scalable_Gaussian_World_Models_for_Robotic_Manipulation_ICCV_2025_paper.html",
    "pdfUrl": "https://arxiv.org/pdf/2508.17600",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@inproceedings{gwm,\n  author       = {Guanxing Lu and\n                  Baoxiong Jia and\n                  Puhao Li and\n                  Yixin Chen and\n                  Ziwei Wang and\n                  Yansong Tang and\n                  Siyuan Huang},\n  title        = {{GWM:} Towards Scalable Gaussian World Models for Robotic Manipulation},\n  booktitle    = {ICCV},\n  pages        = {9263--9274},\n  year         = {2025},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "三维多视角建模",
      "策略后训练与WM-RL"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-17476f447a05757f36fa",
    "title": "X-MOBILITY: End-To-End Generalizable Navigation via World Modeling",
    "authors": "Wei Liu; Huihua Zhao; Chenran Li; Joydeep Biswas; Billy Okal; Pulkit Goyal; Yan Chang; Soha Pouya",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "x_mobility",
    "arxivUrl": null,
    "codeUrls": [
      "https://github.com/NVlabs/X-MOBILITY"
    ],
    "projectUrl": "https://nvlabs.github.io/X-MOBILITY/",
    "venue": "ICRA 2025",
    "paperUrl": "https://ieeexplore.ieee.org/document/11128692/",
    "pdfUrl": "https://arxiv.org/pdf/2410.17491",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@inproceedings{x_mobility,\n  author       = {Wei Liu and\n                  Huihua Zhao and\n                  Chenran Li and\n                  Joydeep Biswas and\n                  Billy Okal and\n                  Pulkit Goyal and\n                  Yan Chang and\n                  Soha Pouya},\n  title        = {{X-MOBILITY:} End-to-End Generalizable Navigation via World Modeling},\n  booktitle    = {ICRA},\n  pages        = {7569--7576},\n  year         = {2025},\n\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "导航",
      "潜空间预测与JEPA"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-0893bf6c64c073787638",
    "title": "OpenVLA: An Open-Source Vision-Language-Action Model",
    "authors": "Moo Jin Kim; Karl Pertsch; Siddharth Karamcheti; Ted Xiao; Ashwin Balakrishna; Suraj Nair; Rafael Rafailov; Ethan Paul Foster; Pannag R. Sanketi; Quan Vuong; Thomas Kollar; Benjamin Burchfiel; Russ Tedrake; Dorsa Sadigh; Sergey Levine; Percy Liang; Chelsea Finn",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "openvla",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "CoRL 2024",
    "paperUrl": "https://proceedings.mlr.press/v270/kim25c.html",
    "pdfUrl": "https://arxiv.org/pdf/2406.09246",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@inproceedings{openvla,\n  author       = {Moo Jin Kim and\n                  Karl Pertsch and\n                  Siddharth Karamcheti and\n                  Ted Xiao and\n                  Ashwin Balakrishna and\n                  Suraj Nair and\n                  Rafael Rafailov and\n                  Ethan Paul Foster and\n                  Pannag R. Sanketi and\n                  Quan Vuong and\n                  Thomas Kollar and\n                  Benjamin Burchfiel and\n                  Russ Tedrake and\n                  Dorsa Sadigh and\n                  Sergey Levine and\n                  Percy Liang and\n                  Chelsea Finn},\n  title        = {OpenVLA: An Open-Source Vision-Language-Action Model},\n  booktitle    = {CoRL},\n  pages        = {2679--2713},\n  year         = {2024},\n\n}",
    "majorCategory": "VLA",
    "subcategories": [
      "自回归VLA"
    ],
    "architecture": "One Model",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-07d6fe5781f45177707e",
    "title": "Driving in the Occupancy World: Vision-Centric 4D Occupancy Forecasting and Planning via World Models for Autonomous Driving",
    "authors": "Yu Yang; Jianbiao Mei; Yukai Ma; Siliang Du; Wenqing Chen; Yijie Qian; Yuxiang Feng; Yong Liu",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "drive_occworld",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "AAAI 2025",
    "paperUrl": "https://ojs.aaai.org/index.php/AAAI/article/view/33010",
    "pdfUrl": "https://ojs.aaai.org/index.php/AAAI/article/download/33010/35165",
    "doi": "https://doi.org/10.1609/aaai.v39i9.33010",
    "publicationYear": 2025,
    "bibtex": "@inproceedings{drive_occworld,\n  author       = {Yu Yang and\n                  Jianbiao Mei and\n                  Yukai Ma and\n                  Siliang Du and\n                  Wenqing Chen and\n                  Yijie Qian and\n                  Yuxiang Feng and\n                  Yong Liu},\n  title        = {Driving in the Occupancy World: Vision-Centric 4D Occupancy Forecasting\n                  and Planning via World Models for Autonomous Driving},\n  booktitle    = {AAAI},\n  pages        = {9327--9335},\n  year         = {2025},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "自动驾驶",
      "三维多视角建模"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2506.19842",
    "title": "ManiGaussian++: General Robotic Bimanual Manipulation with Hierarchical Gaussian World Model",
    "authors": "Tengbo Yu; Guanxing Lu; Zaijia Yang; Haoyuan Deng; Season Si Chen; Jiwen Lu; Wenbo Ding; Guoqiang Hu; Yansong Tang; Ziwei Wang",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "manigauss_pp",
    "arxivUrl": "https://arxiv.org/abs/2506.19842",
    "codeUrls": [
      "https://github.com/April-Yz/ManiGaussian_Bimanual"
    ],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2506.19842",
    "pdfUrl": "https://arxiv.org/pdf/2506.19842",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@inproceedings{manigauss_pp,\n  author       = {Tengbo Yu and\n                  Guanxing Lu and\n                  Zaijia Yang and\n                  Haoyuan Deng and\n                  Season Si Chen and\n                  Jiwen Lu and\n                  Wenbo Ding and\n                  Guoqiang Hu and\n                  Yansong Tang and\n                  Ziwei Wang},\n  title        = {ManiGaussian++: General Robotic Bimanual Manipulation with Hierarchical\n                  Gaussian World Model},\n  booktitle    = {IROS},\n  pages        = {12232--12239},\n  year         = {2025},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "三维多视角建模",
      "泛化与动作对齐"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2503.06669",
    "title": "AgiBot World Colosseo: A Large-scale Manipulation Platform for Scalable and Intelligent Embodied Systems",
    "authors": "AgiBot-World-Contributors; Qingwen Bu; Jisong Cai; Li Chen; Xiuqi Cui; Yan Ding; Siyuan Feng; Shenyuan Gao; Xindong He; Xuan Hu; Xu Huang; Shu Jiang; Yuxin Jiang; Cheng Jing; Hongyang Li; Jialu Li; Chiming Liu; Yi Liu; Yuxiang Lu; Jianlan Luo; Ping Luo; Yao Mu; Yuehan Niu; Yixuan Pan; Jiangmiao Pang; Yu Qiao; Guanghui Ren; Cheng Ruan; Jiaqi Shan; Yongjian Shen; Chengshi Shi; Mingkang Shi; Modi Shi; Chonghao Sima; Jianheng Song; Huijie Wang; Wenhao Wang; Dafeng Wei; Chengen Xie; Guo Xu; Junchi Yan; Cunbiao Yang; Lei Yang; Shukai Yang; Maoqing Yao; Jia Zeng; Chi Zhang; Qinglin Zhang; Bin Zhao; Chengyue Zhao; Jiaqi Zhao; Jianchao Zhu",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "agibotworld",
    "arxivUrl": "https://arxiv.org/abs/2503.06669",
    "codeUrls": [
      "https://github.com/OpenDriveLab/AgiBot-World"
    ],
    "projectUrl": "https://agibot-world.com/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2503.06669",
    "pdfUrl": "https://arxiv.org/pdf/2503.06669",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@inproceedings{agibotworld,\n  title={Agibot world colosseo: A large-scale manipulation platform for scalable and intelligent embodied systems},\n  author={Bu, Qingwen and Cai, Jisong and Chen, Li and Cui, Xiuqi and Ding, Yan and Feng, Siyuan and He, Xindong and Huang, Xu and others},\n  booktitle={IROS},\n  year={2025},\n}",
    "majorCategory": "数据集",
    "subcategories": [
      "机器人示范与操作数据",
      "跨机器人与多任务数据",
      "数据采集接口"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2502.00622",
    "title": "Inference-Time Enhancement of Generative Robot Policies via Predictive World Modeling",
    "authors": "Han Qi; Haocheng Yin; Aris Zhu; Yilun Du; Heng Yang",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "gpc",
    "arxivUrl": "https://arxiv.org/abs/2502.00622",
    "codeUrls": [
      "https://github.com/han20192019/gpc_code"
    ],
    "projectUrl": "https://computationalrobotics.seas.harvard.edu/GPC/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2502.00622",
    "pdfUrl": "https://arxiv.org/pdf/2502.00622",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@article{gpc,\n  author       = {Han Qi and\n                  Haocheng Yin and\n                  Aris Zhu and\n                  Yilun Du and\n                  Heng Yang},\n  title        = {Inference-Time Enhancement of Generative Robot Policies via Predictive\n                  World Modeling},\n  journal      = {RA-L},\n  volume       = {11},\n  number       = {5},\n  pages        = {5534--5541},\n  year         = {2026},\n\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "策略后训练与WM-RL"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2406.08481",
    "title": "Enhancing End-to-End Autonomous Driving with Latent World Model",
    "authors": "Yingyan Li; Lue Fan; Jiawei He; Yuqi Wang; Yuntao Chen; Zhaoxiang Zhang; Tieniu Tan",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "enhance_e2e_ad",
    "arxivUrl": "https://arxiv.org/abs/2406.08481",
    "codeUrls": [
      "https://github.com/BraveGroup/LAW"
    ],
    "projectUrl": null,
    "venue": "ICLR 2025",
    "paperUrl": "https://arxiv.org/abs/2406.08481",
    "pdfUrl": "https://arxiv.org/pdf/2406.08481",
    "doi": null,
    "publicationYear": 2025,
    "bibtex": "@inproceedings{enhance_e2e_ad,\n  author       = {Yingyan Li and\n                  Lue Fan and\n                  Jiawei He and\n                  Yuqi Wang and\n                  Yuntao Chen and\n                  Zhaoxiang Zhang and\n                  Tieniu Tan},\n  title        = {Enhancing End-to-End Autonomous Driving with Latent World Model},\n  booktitle    = {ICLR},\n  year         = {2025},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "自动驾驶",
      "潜空间预测与JEPA"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2412.09627",
    "title": "Doe-1: Closed-Loop Autonomous Driving with Large World Model",
    "authors": "Wenzhao Zheng; Zetian Xia; Yuanhui Huang; Sicheng Zuo; Jie Zhou; Jiwen Lu",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2024-12-12",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "doe_1",
    "arxivUrl": "https://arxiv.org/abs/2412.09627",
    "codeUrls": [
      "https://github.com/wzzheng/Doe"
    ],
    "projectUrl": "https://wzzheng.net/Doe",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2412.09627",
    "pdfUrl": "https://arxiv.org/pdf/2412.09627",
    "doi": null,
    "publicationYear": 2024,
    "bibtex": "@article{doe_1,\n      title={Doe-1: Closed-Loop Autonomous Driving with Large World Model}, \n      author={Wenzhao Zheng and Zetian Xia and Yuanhui Huang and Sicheng Zuo and Jie Zhou and Jiwen Lu},\n      year={2024},\n      journal={arXiv:2412.09627},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "联合视频动作建模",
      "自动驾驶"
    ],
    "architecture": "One Model",
    "predictionParadigm": "联合预测",
    "quadrant": "Q1 · One Model × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2410.06158",
    "title": "GR-2: A Generative Video-Language-Action Model with Web-Scale Knowledge for Robot Manipulation",
    "authors": "Chi-Lam Cheang; Guangzeng Chen; Ya Jing; Tao Kong; Hang Li; Yifeng Li; Yuxiao Liu; Hongtao Wu; Jiafeng Xu; Yichu Yang; Hanbo Zhang; Minzhao Zhu",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2024-10-08",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "gr2",
    "arxivUrl": "https://arxiv.org/abs/2410.06158",
    "codeUrls": [],
    "projectUrl": "https://gr2-manipulation.github.io",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2410.06158",
    "pdfUrl": "https://arxiv.org/pdf/2410.06158",
    "doi": null,
    "publicationYear": 2024,
    "bibtex": "@article{gr2,\n      title={GR-2: A Generative Video-Language-Action Model with Web-Scale Knowledge for Robot Manipulation}, \n      author={Chi-Lam Cheang and Guangzeng Chen and Ya Jing and Tao Kong and Hang Li and Yifeng Li and Yuxiao Liu and Hongtao Wu and Jiafeng Xu and Yichu Yang and Hanbo Zhang and Minzhao Zhu},\n      year={2024},\n      journal={arXiv:2410.06158},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "联合视频动作建模",
      "泛化与动作对齐"
    ],
    "architecture": "One Model",
    "predictionParadigm": "联合预测",
    "quadrant": "Q1 · One Model × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2410.00564",
    "title": "Scaling Offline Model-Based RL via Jointly-Optimized World-Action Model Pretraining",
    "authors": "Jie Cheng; Ruixi Qiao; Yingwei Ma; Binhua Li; Gang Xiong; Qinghai Miao; Yongbin Li; Yisheng Lv",
    "affiliations": "State Key Laboratory of Multimodal Artificial Intelligence Systems, Institute of Automation, Chinese Academy of Sciences; School of Artificial Intelligence, University of Chinese Academy of Sciences; Alibaba Group",
    "contribution": "JOWA learns an Atari world model and distributional Q-function through one shared transformer, then searches short imagined futures to choose actions. Its strongest evidence is improved aggregate game return and data-efficient offline adaptation; neither universal game-wise scaling nor unconditional planning optimality is established (e02, e03, e04, e07, e08, e11, e12).",
    "abstract": "",
    "submittedDate": "2024-10-01",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv241000564",
    "arxivUrl": "https://arxiv.org/abs/2410.00564",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2410.00564",
    "pdfUrl": "https://arxiv.org/pdf/2410.00564",
    "doi": null,
    "publicationYear": 2024,
    "bibtex": "",
    "majorCategory": "奠基性工作",
    "subcategories": [
      "经典WM与模型式RL"
    ],
    "architecture": null,
    "predictionParadigm": null,
    "quadrant": null,
    "classificationStatus": null
  },
  {
    "id": "2409.16283",
    "title": "Gen2Act: Human Video Generation in Novel Scenarios enables Generalizable Robot Manipulation",
    "authors": "Homanga Bharadhwaj; Debidatta Dwibedi; Abhinav Gupta; Shubham Tulsiani; Carl Doersch; Ted Xiao; Dhruv Shah; Fei Xia; Dorsa Sadigh; Sean Kirmani",
    "affiliations": "Google DeepMind; The Robotics Institute, Carnegie Mellon University; Computer Science Department, Stanford University",
    "contribution": "Gen2Act turns a language instruction and an initial scene image into a generated human demonstration, then uses that video to condition a separate closed-loop robot policy. A pretrained VideoPoet supplies the demonstration without robot-specific fine-tuning. Auxiliary point-track prediction teaches policy representations to retain motion cues, while deployment predicts actions directly from video features and recent robot observations. Real robot results support improved generalization relative to the reported baselines, but plausible generation does not ensure correct execution, and long-horizon reliability remains limited.",
    "abstract": "",
    "submittedDate": "2024-09-24",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv240916283",
    "arxivUrl": "https://arxiv.org/abs/2409.16283",
    "codeUrls": [],
    "projectUrl": null,
    "venue": "arXiv preprint",
    "paperUrl": "https://arxiv.org/abs/2409.16283",
    "pdfUrl": "https://arxiv.org/pdf/2409.16283",
    "doi": null,
    "publicationYear": 2024,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "视觉规划与IDM",
      "泛化与动作对齐"
    ],
    "architecture": "待核实",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": null
  },
  {
    "id": "2409.12192",
    "title": "DynaMo: In-Domain Dynamics Pretraining for Visuo-Motor Control",
    "authors": "Zichen Jeff Cui; Hengkai Pan; Aadhithya Iyer; Siddhant Haldar; Lerrel Pinto",
    "affiliations": "New York University",
    "contribution": "DynaMo uses action-free dynamics pretraining to make visual features useful for imitation learning. An encoder, latent inverse model and forward model jointly learn to predict next-frame embeddings; a separate policy then learns from frozen features and labeled demonstrations. Results favor DynaMo on several manipulation tasks, but include ties and initialization regressions. The contribution is a representation-learning objective, with no demonstrated use of the dynamics models for online planning.",
    "abstract": "",
    "submittedDate": "2024-09-18",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv240912192",
    "arxivUrl": "https://arxiv.org/abs/2409.12192",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2409.12192",
    "pdfUrl": "https://arxiv.org/pdf/2409.12192",
    "doi": null,
    "publicationYear": 2024,
    "bibtex": "",
    "majorCategory": "Related resources",
    "subcategories": [
      "Latent-action & representation methods"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": null
  },
  {
    "id": "2409.03272",
    "title": "OccLLaMA: An Occupancy-Language-Action Generative World Model for Autonomous Driving",
    "authors": "Julong Wei; Shanshuai Yuan; Pengfei Li; Qingda Hu; Zhongxue Gan; Wenchao Ding",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2024-09-05",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "occllama",
    "arxivUrl": "https://arxiv.org/abs/2409.03272",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2409.03272",
    "pdfUrl": "https://arxiv.org/pdf/2409.03272",
    "doi": null,
    "publicationYear": 2024,
    "bibtex": "@article{occllama,\n      title={Occllama: An occupancy-language-action generative world model for autonomous driving}, \n      author={Wei, Julong and Yuan, Shanshuai and Li, Pengfei and Hu, Qingda and Gan, Zhongxue and Ding, Wenchao},\n      year={2024},\n      journal={arXiv:2409.03272},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "联合视频动作建模",
      "三维多视角建模",
      "自动驾驶"
    ],
    "architecture": "One Model",
    "predictionParadigm": "联合预测",
    "quadrant": "Q1 · One Model × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2407.07726",
    "title": "PaliGemma: A versatile 3B VLM for transfer",
    "authors": "Lucas Beyer; Andreas Steiner; André Susano Pinto; Alexander Kolesnikov; Xiao Wang; Daniel Salz; Maxim Neumann; Ibrahim Alabdulmohsin; Michael Tschannen; Emanuele Bugliarello; Thomas Unterthiner; Daniel Keysers; Skanda Koppula; Fangyu Liu; Adam Grycner; Alexey Gritsenko; Neil Houlsby; Manoj Kumar; Keran Rong; Julian Eisenschlos; Rishabh Kabra; Matthias Bauer; Matko Bošnjak; Xi Chen; Matthias Minderer; Paul Voigtlaender; Ioana Bica; Ivana Balazevic; Joan Puigcerver; Pinelopi Papalampidi; Olivier Henaff; Xi Xiong; Radu Soricut; Jeremiah Harmsen; Xiaohua Zhai",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2024-07-10",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "paligemma",
    "arxivUrl": "https://arxiv.org/abs/2407.07726",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2407.07726",
    "pdfUrl": "https://arxiv.org/pdf/2407.07726",
    "doi": null,
    "publicationYear": 2024,
    "bibtex": "@article{paligemma,\n      title={PaliGemma: A versatile 3B VLM for transfer}, \n      author={Lucas Beyer and Andreas Steiner and André Susano Pinto and Alexander Kolesnikov and Xiao Wang and Daniel Salz and Maxim Neumann and Ibrahim Alabdulmohsin and Michael Tschannen and Emanuele Bugliarello and Thomas Unterthiner and Daniel Keysers and Skanda Koppula and Fangyu Liu and Adam Grycner and Alexey Gritsenko and Neil Houlsby and Manoj Kumar and Keran Rong and Julian Eisenschlos and Rishabh Kabra and Matthias Bauer and Matko Bošnjak and Xi Chen and Matthias Minderer and Paul Voigtlaender and Ioana Bica and Ivana Balazevic and Joan Puigcerver and Pinelopi Papalampidi and Olivier Henaff and Xi Xiong and Radu Soricut and Jeremiah Harmsen and Xiaohua Zhai},\n      year={2024},\n      journal={arXiv:2407.07726},\n}",
    "majorCategory": "WAM Components",
    "subcategories": [
      "Language & vision-language backbones",
      "语言与VLM Backbone"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2407.05530",
    "title": "This&That: Language-Gesture Controlled Video Generation for Robot Planning",
    "authors": "Boyang Wang; Nikhil Sridhar; Chao Feng; Mark Van der Merwe; Adam Fishman; Nima Fazeli; Jeong Joon Park",
    "affiliations": "University of Michigan; University of Washington",
    "contribution": "This&That turns an initial image, language and pointing coordinates into a generated video plan, then uses a separate DiVA behavior-cloning controller to follow it with live feedback. Gestures improve intent disambiguation in Bridge video generation and simulated block manipulation; real-robot execution remains untested.",
    "abstract": "",
    "submittedDate": "2024-07-08",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv240705530",
    "arxivUrl": "https://arxiv.org/abs/2407.05530",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2407.05530",
    "pdfUrl": "https://arxiv.org/pdf/2407.05530",
    "doi": null,
    "publicationYear": 2024,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "视觉规划与IDM"
    ],
    "architecture": "待核实",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": null
  },
  {
    "id": "2406.16862",
    "title": "Dreamitate: Real-World Visuomotor Policy Learning via Video Generation",
    "authors": "Junbang Liang; Ruoshi Liu; Ege Ozguroglu; Sruthi Sudhakar; Achal Dave; Pavel Tokmakov; Shuran Song; Carl Vondrick",
    "affiliations": "Columbia University; Toyota Research Institute; Stanford University",
    "contribution": "Dreamitate learns task-specific video generation from human tool demonstrations, then extracts tool poses from synthesized stereo videos for robot execution. It outperforms the tested Diffusion Policy baseline on four physical manipulation tasks, while depending on calibrated cameras, known rigid tools, and slow open-loop generation. The central evidence concerns executed tool trajectories under specified object and scene shifts, rather than unrestricted manipulation or a general action-conditioned simulator.",
    "abstract": "",
    "submittedDate": "2024-06-24",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv240616862",
    "arxivUrl": "https://arxiv.org/abs/2406.16862",
    "codeUrls": [],
    "projectUrl": null,
    "venue": "arXiv preprint",
    "paperUrl": "https://arxiv.org/abs/2406.16862",
    "pdfUrl": "https://arxiv.org/pdf/2406.16862",
    "doi": null,
    "publicationYear": 2024,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "视觉规划与IDM"
    ],
    "architecture": "待核实",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": null
  },
  {
    "id": "2404.12377",
    "title": " RoboDreamer: Learning Compositional World Models for Robot Imagination",
    "authors": "Siyuan Zhou; Yilun Du; Jiaben Chen; Yandong Li; Dit-Yan Yeung; Chuang Gan",
    "affiliations": "Hong Kong University of Science and Technology; Massachusetts Institute of Technology; University of California, San Diego; Google Research; University of Massachusetts Amherst; MIT-IBM Watson AI Lab",
    "contribution": "RoboDreamer composes phrase-conditioned video diffusion predictions to imagine robot plans for unfamiliar instruction combinations. Optional goal images or sketches sharpen spatial specifications; a separate inverse-dynamics model converts imagined frames into actions. The strongest language-only evidence concerns human-rated video alignment, while executed success is measured separately in RLBench simulation (E03, E10–E13).",
    "abstract": "",
    "submittedDate": "2024-04-18",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv240412377",
    "arxivUrl": "https://arxiv.org/abs/2404.12377",
    "codeUrls": [],
    "projectUrl": null,
    "venue": "arXiv preprint",
    "paperUrl": "https://arxiv.org/abs/2404.12377",
    "pdfUrl": "https://arxiv.org/pdf/2404.12377",
    "doi": null,
    "publicationYear": 2024,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "视觉规划与IDM"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": null
  },
  {
    "id": "ref-eecf2578168f85db0c45",
    "title": "Prediction with Action: Visual Policy Learning via Joint Denoising Process",
    "authors": "Yanjiang Guo; Yucheng Hu; Jianke Zhang; Yen-Jen Wang; Xiaoyu Chen; Chaochao Lu; Jianyu Chen",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "padothers; pad",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "NeurIPS 2024",
    "paperUrl": "https://proceedings.neurips.cc/paper_files/paper/2024/hash/cbe25fa0e7c7084049276888a09acc8d-Abstract-Conference.html",
    "pdfUrl": "https://arxiv.org/pdf/2411.18179",
    "doi": "https://doi.org/10.52202/079017-3570",
    "publicationYear": 2024,
    "bibtex": "@inproceedings{padothers,\n  author       = {Yanjiang Guo and\n                  Yucheng Hu and\n                  Jianke Zhang and\n                  Yen{-}Jen Wang and\n                  Xiaoyu Chen and\n                  Chaochao Lu and\n                  Jianyu Chen},\n  title        = {Prediction with Action: Visual Policy Learning via Joint Denoising\n                  Process},\n  booktitle    = {NeurIPS},\n  year         = {2024},\n}\n\n@article{pad,\n  title={Prediction with action: Visual policy learning via joint denoising process},\n  author={Guo, Yanjiang and Hu, Yucheng and Zhang, Jianke and Wang, Yen-Jen and Chen, Xiaoyu and Lu, Chaochao and Chen, Jianyu},\n  journal={NeurIPS},\n  volume={37},\n  pages={112386--112410},\n  year={2024}\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "联合视频动作建模"
    ],
    "architecture": "One Model",
    "predictionParadigm": "联合预测",
    "quadrant": "Q1 · One Model × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-e25271fa6f028e5611cf",
    "title": "NAVSIM: Data-Driven Non-Reactive Autonomous Vehicle Simulation and Benchmarking",
    "authors": "Daniel Dauner; Marcel Hallgarten; Tianyu Li; Xinshuo Weng; Zhiyu Huang; Zetong Yang; Hongyang Li; Igor Gilitschenski; Boris Ivanovic; Marco Pavone; Andreas Geiger; Kashyap Chitta",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "navsim",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "NeurIPS 2024 Datasets and Benchmarks Track",
    "paperUrl": "https://proceedings.neurips.cc/paper_files/paper/2024/file/32768f7faf1995026ef9821c696f3404-Paper-Datasets_and_Benchmarks_Track.pdf",
    "pdfUrl": "https://proceedings.neurips.cc/paper_files/paper/2024/file/32768f7faf1995026ef9821c696f3404-Paper-Datasets_and_Benchmarks_Track.pdf",
    "doi": null,
    "publicationYear": 2024,
    "bibtex": "@inproceedings{navsim,\n  author       = {Daniel Dauner and\n                  Marcel Hallgarten and\n                  Tianyu Li and\n                  Xinshuo Weng and\n                  Zhiyu Huang and\n                  Zetong Yang and\n                  Hongyang Li and\n                  Igor Gilitschenski and\n                  Boris Ivanovic and\n                  Marco Pavone and\n                  Andreas Geiger and\n                  Kashyap Chitta},\n  title        = {{NAVSIM:} Data-Driven Non-Reactive Autonomous Vehicle Simulation and\n                  Benchmarking},\n  booktitle    = {NeurIPS},\n  year         = {2024},\n}",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "自动驾驶基准",
      "评测协议与诊断",
      "驾驶仿真"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-e1a2abbaffcea1e2e971",
    "title": "Unleashing Large-Scale Video Generative Pre-training for Visual Robot Manipulation",
    "authors": "Hongtao Wu; Ya Jing; Chilam Cheang; Guangzeng Chen; Jiafeng Xu; Xinghang Li; Minghuan Liu; Hang Li; Tao Kong",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "gr1",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "ICLR 2024",
    "paperUrl": "https://proceedings.iclr.cc/paper_files/paper/2024/hash/2c37c5bcef24b9541550261dcd63261b-Abstract-Conference.html",
    "pdfUrl": "https://proceedings.iclr.cc/paper_files/paper/2024/file/2c37c5bcef24b9541550261dcd63261b-Paper-Conference.pdf",
    "doi": null,
    "publicationYear": 2024,
    "bibtex": "@inproceedings{gr1,\n  author       = {Hongtao Wu and\n                  Ya Jing and\n                  Chilam Cheang and\n                  Guangzeng Chen and\n                  Jiafeng Xu and\n                  Xinghang Li and\n                  Minghuan Liu and\n                  Hang Li and\n                  Tao Kong},\n  title        = {Unleashing Large-Scale Video Generative Pre-training for Visual Robot\n                  Manipulation},\n  booktitle    = {ICLR},\n  year         = {2024},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "联合视频动作建模"
    ],
    "architecture": "One Model",
    "predictionParadigm": "联合预测",
    "quadrant": "Q1 · One Model × 联合预测",
    "classificationStatus": "综述明确"
  },
  {
    "id": "ref-e18cd2f0df3768294977",
    "title": "DINOv2: Learning Robust Visual Features without Supervision",
    "authors": "Maxime Oquab; Timothée Darcet; Théo Moutakanni; Huy V. Vo; Marc Szafraniec; Vasil Khalidov; Pierre Fernandez; Daniel Haziza; Francisco Massa; Alaaeldin El-Nouby; Mido Assran; Nicolas Ballas; Wojciech Galuba; Russell Howes; Po-Yao Huang; Shang-Wen Li; Ishan Misra; Michael Rabbat; Vasu Sharma; Gabriel Synnaeve; Hu Xu; Hervé Jégou; Julien Mairal; Patrick Labatut; Armand Joulin; Piotr Bojanowski",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "dinov2",
    "arxivUrl": null,
    "codeUrls": [
      "https://github.com/facebookresearch/dinov2"
    ],
    "projectUrl": null,
    "venue": "Transactions on Machine Learning Research",
    "paperUrl": "https://openreview.net/forum?id=a68SUt6zFt",
    "pdfUrl": "https://arxiv.org/pdf/2304.07193",
    "doi": null,
    "publicationYear": 2024,
    "bibtex": "@article{dinov2,\n  author       = {Maxime Oquab and\n                  Timoth{\\'{e}}e Darcet and\n                  Th{\\'{e}}o Moutakanni and\n                  Huy V. Vo and\n                  Marc Szafraniec and\n                  Vasil Khalidov and\n                  Pierre Fernandez and\n                  Daniel Haziza and\n                  Francisco Massa and\n                  Alaaeldin El{-}Nouby and\n                  Mido Assran and\n                  Nicolas Ballas and\n                  Wojciech Galuba and\n                  Russell Howes and\n                  Po{-}Yao Huang and\n                  Shang{-}Wen Li and\n                  Ishan Misra and\n                  Michael Rabbat and\n                  Vasu Sharma and\n                  Gabriel Synnaeve and\n                  Hu Xu and\n                  Herv{\\'{e}} J{\\'{e}}gou and\n                  Julien Mairal and\n                  Patrick Labatut and\n                  Armand Joulin and\n                  Piotr Bojanowski},\n  title        = {DINOv2: Learning Robust Visual Features without Supervision},\n  journal      = {TMLR},\n  year         = {2024},\n  \n}",
    "majorCategory": "WAM Components",
    "subcategories": [
      "视觉编码器与表征"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-e064f89fc62c8df5da9f",
    "title": "Universal Manipulation Interface: In-The-Wild Robot Teaching Without In-The-Wild Robots",
    "authors": "Cheng Chi; Zhenjia Xu; Chuer Pan; Eric Cousineau; Benjamin Burchfiel; Siyuan Feng; Russ Tedrake; Shuran Song",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "UMI",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "Robotics: Science and Systems (RSS) 2024",
    "paperUrl": "https://www.roboticsproceedings.org/rss20/p045.html",
    "pdfUrl": "http://www.roboticsproceedings.org/rss20/p045.pdf",
    "doi": null,
    "publicationYear": 2024,
    "bibtex": "@inproceedings{UMI,\n  author       = {Cheng Chi and\n                  Zhenjia Xu and\n                  Chuer Pan and\n                  Eric Cousineau and\n                  Benjamin Burchfiel and\n                  Siyuan Feng and\n                  Russ Tedrake and\n                  Shuran Song},\n  title        = {Universal Manipulation Interface: In-The-Wild Robot Teaching Without\n                  In-The-Wild Robots},\n  booktitle    = {RSS},\n  year         = {2024},\n\n}",
    "majorCategory": "数据集",
    "subcategories": [
      "数据采集接口",
      "机器人示范与操作数据",
      "跨机器人与多任务数据"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "综述明确"
  },
  {
    "id": "ref-da609a82e0cab3355273",
    "title": "World Models via Policy-Guided Trajectory Diffusion",
    "authors": "Marc Rigter; Jun Yamada; Ingmar Posner",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "polygrad",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "TMLR 2024",
    "paperUrl": "https://openreview.net/forum?id=9CcgO0LhKG",
    "pdfUrl": "https://arxiv.org/pdf/2312.08533",
    "doi": null,
    "publicationYear": 2024,
    "bibtex": "@article{polygrad,\n  author       = {Marc Rigter and\n                  Jun Yamada and\n                  Ingmar Posner},\n  title        = {World Models via Policy-Guided Trajectory Diffusion},\n  journal      = {TMLR},\n  volume       = {2024},\n  year         = {2024},\n}",
    "majorCategory": "奠基性工作",
    "subcategories": [
      "经典WM与模型式RL",
      "扩散与流匹配基础"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-c4a542b17ebebe61f9ec",
    "title": "Bench2Drive: Towards Multi-Ability Benchmarking of Closed-Loop End-To-End Autonomous Driving",
    "authors": "Xiaosong Jia; Zhenjie Yang; Qifeng Li; Zhiyuan Zhang; Junchi Yan",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "bench2drive",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "NeurIPS 2024 Datasets and Benchmarks Track",
    "paperUrl": "https://proceedings.neurips.cc/paper_files/paper/2024/hash/017761f94a1cd66d01c041aff85492c4-Abstract-Datasets_and_Benchmarks_Track.html",
    "pdfUrl": "https://arxiv.org/pdf/2406.03877",
    "doi": null,
    "publicationYear": 2024,
    "bibtex": "@inproceedings{bench2drive,\n  author       = {Xiaosong Jia and\n                  Zhenjie Yang and\n                  Qifeng Li and\n                  Zhiyuan Zhang and\n                  Junchi Yan},\n  title        = {Bench2Drive: Towards Multi-Ability Benchmarking of Closed-Loop End-To-End\n                  Autonomous Driving},\n  booktitle    = {NeurIPS},\n  year         = {2024},\n}",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "自动驾驶基准",
      "物理仿真",
      "评测协议与诊断"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-baf06f1a599804c753fb",
    "title": "Generalized Predictive Model for Autonomous Driving",
    "authors": "Jiazhi Yang; Shenyuan Gao; Yihang Qiu; Li Chen; Tianyu Li; Bo Dai; Kashyap Chitta; Penghao Wu; Jia Zeng; Ping Luo; Jun Zhang; Andreas Geiger; Yu Qiao; Hongyang Li",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "opendv",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "CVPR 2024",
    "paperUrl": "https://openaccess.thecvf.com/content/CVPR2024/html/Yang_Generalized_Predictive_Model_for_Autonomous_Driving_CVPR_2024_paper.html",
    "pdfUrl": "https://openaccess.thecvf.com/content/CVPR2024/papers/Yang_Generalized_Predictive_Model_for_Autonomous_Driving_CVPR_2024_paper.pdf",
    "doi": null,
    "publicationYear": 2024,
    "bibtex": "@inproceedings{opendv,\n  author       = {Jiazhi Yang and\n                  Shenyuan Gao and\n                  Yihang Qiu and\n                  Li Chen and\n                  Tianyu Li and\n                  Bo Dai and\n                  Kashyap Chitta and\n                  Penghao Wu and\n                  Jia Zeng and\n                  Ping Luo and\n                  Jun Zhang and\n                  Andreas Geiger and\n                  Yu Qiao and\n                  Hongyang Li},\n  title        = {Generalized Predictive Model for Autonomous Driving},\n  booktitle    = {CVPR},\n  pages        = {14662--14672},\n  year         = {2024},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "神经世界模拟器"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-b116da337363048621f5",
    "title": "State Estimation for Robotics",
    "authors": "Timothy D. Barfoot",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "barfoot2024state",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "Cambridge University Press (Book, 2nd edition)",
    "paperUrl": "https://www.cambridge.org/core/books/state-estimation-for-robotics/00E53274A2F1E6CC1A55CA5C3D1C9718",
    "pdfUrl": null,
    "doi": "https://doi.org/10.1017/9781009299909",
    "publicationYear": 2024,
    "bibtex": "@book{barfoot2024state,\n  title={State estimation for robotics},\n  author={Barfoot, Timothy D},\n  year={2024},\n  publisher={Cambridge University Press}\n}",
    "majorCategory": "奠基性工作",
    "subcategories": [
      "三维表示与状态估计",
      "理论与规划"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "综述明确"
  },
  {
    "id": "ref-aa9c28fbf4242ea03696",
    "title": "DROID: A Large-Scale In-The-Wild Robot Manipulation Dataset",
    "authors": "Alexander Khazatsky; Karl Pertsch; Suraj Nair; Ashwin Balakrishna; Sudeep Dasari; Siddharth Karamcheti; Soroush Nasiriany; Mohan Kumar Srirama; Lawrence Yunliang Chen; Kirsty Ellis; Peter David Fagan; Joey Hejna; Masha Itkina; Marion Lepert; Yecheng Jason Ma; Patrick Tree Miller; Jimmy Wu; Suneel Belkhale; Shivin Dass; Huy Ha; Arhan Jain; Abraham Lee; Youngwoon Lee; Marius Memmel; Sungjae Park; Ilija Radosavovic; Kaiyuan Wang; Albert Zhan; Kevin Black; Cheng Chi; Kyle Beltran Hatch; Shan Lin; Jingpei Lu; Jean Mercat; Abdul Rehman; Pannag R. Sanketi; Archit Sharma; Cody Simpson; Quan Vuong; Homer Rich Walke; Blake Wulfe; Ted Xiao; Jonathan Heewon Yang; Arefeh Yavary; Tony Z. Zhao; Christopher Agia; Rohan Baijal; Mateo Guaman Castro; Daphne Chen; Qiuyu Chen; Trinity Chung; Jaimyn Drake; Ethan Paul Foster; Jensen Gao; David Antonio Herrera; Minho Heo; Kyle Hsu; Jiaheng Hu; Donovon Jackson; Charlotte Le; Yunshuang Li; Roy Lin; Zehan Ma; Abhiram Maddukuri; Suvir Mirchandani; Daniel Morton; Tony Nguyen; Abigail O'Neill; Rosario Scalise; Derick Seale; Victor Son; Stephen Tian; Emi Tran; Andrew E. Wang; Yilin Wu; Annie Xie; Jingyun Yang; Patrick Yin; Yunchu Zhang; Osbert Bastani; Glen Berseth; Jeannette Bohg; Ken Goldberg; Abhinav Gupta; Abhishek Gupta; Dinesh Jayaraman; Joseph J. Lim; Jitendra Malik; Roberto Mart\\'ın-Mart\\'ın; Subramanian Ramamoorthy; Dorsa Sadigh; Shuran Song; Jiajun Wu; Michael C. Yip; Yuke Zhu; Thomas Kollar; Sergey Levine; Chelsea Finn",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "droid",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "RSS 2024",
    "paperUrl": "https://www.roboticsproceedings.org/rss20/p120.pdf",
    "pdfUrl": "https://www.roboticsproceedings.org/rss20/p120.pdf",
    "doi": null,
    "publicationYear": 2024,
    "bibtex": "@inproceedings{droid,\n  author       = {Alexander Khazatsky and\n                  Karl Pertsch and\n                  Suraj Nair and\n                  Ashwin Balakrishna and\n                  Sudeep Dasari and\n                  Siddharth Karamcheti and\n                  Soroush Nasiriany and\n                  Mohan Kumar Srirama and\n                  Lawrence Yunliang Chen and\n                  Kirsty Ellis and\n                  Peter David Fagan and\n                  Joey Hejna and\n                  Masha Itkina and\n                  Marion Lepert and\n                  Yecheng Jason Ma and\n                  Patrick Tree Miller and\n                  Jimmy Wu and\n                  Suneel Belkhale and\n                  Shivin Dass and\n                  Huy Ha and\n                  Arhan Jain and\n                  Abraham Lee and\n                  Youngwoon Lee and\n                  Marius Memmel and\n                  Sungjae Park and\n                  Ilija Radosavovic and\n                  Kaiyuan Wang and\n                  Albert Zhan and\n                  Kevin Black and\n                  Cheng Chi and\n                  Kyle Beltran Hatch and\n                  Shan Lin and\n                  Jingpei Lu and\n                  Jean Mercat and\n                  Abdul Rehman and\n                  Pannag R. Sanketi and\n                  Archit Sharma and\n                  Cody Simpson and\n                  Quan Vuong and\n                  Homer Rich Walke and\n                  Blake Wulfe and\n                  Ted Xiao and\n                  Jonathan Heewon Yang and\n                  Arefeh Yavary and\n                  Tony Z. Zhao and\n                  Christopher Agia and\n                  Rohan Baijal and\n                  Mateo Guaman Castro and\n                  Daphne Chen and\n                  Qiuyu Chen and\n                  Trinity Chung and\n                  Jaimyn Drake and\n                  Ethan Paul Foster and\n                  Jensen Gao and\n                  David Antonio Herrera and\n                  Minho Heo and\n                  Kyle Hsu and\n                  Jiaheng Hu and\n                  Donovon Jackson and\n                  Charlotte Le and\n                  Yunshuang Li and\n                  Roy Lin and\n                  Zehan Ma and\n                  Abhiram Maddukuri and\n                  Suvir Mirchandani and\n                  Daniel Morton and\n                  Tony Nguyen and\n                  Abigail O'Neill and\n                  Rosario Scalise and\n                  Derick Seale and\n                  Victor Son and\n                  Stephen Tian and\n                  Emi Tran and\n                  Andrew E. Wang and\n                  Yilin Wu and\n                  Annie Xie and\n                  Jingyun Yang and\n                  Patrick Yin and\n                  Yunchu Zhang and\n                  Osbert Bastani and\n                  Glen Berseth and\n                  Jeannette Bohg and\n                  Ken Goldberg and\n                  Abhinav Gupta and\n                  Abhishek Gupta and\n                  Dinesh Jayaraman and\n                  Joseph J. Lim and\n                  Jitendra Malik and\n                  Roberto Mart{\\'{\\i}}n{-}Mart{\\'{\\i}}n and\n                  Subramanian Ramamoorthy and\n                  Dorsa Sadigh and\n                  Shuran Song and\n                  Jiajun Wu and\n                  Michael C. Yip and\n                  Yuke Zhu and\n                  Thomas Kollar and\n                  Sergey Levine and\n                  Chelsea Finn},\n  title        = {{DROID:} {A} Large-Scale In-The-Wild Robot Manipulation Dataset},\n  booktitle    = {RSS},\n  year         = {2024},\n\n}",
    "majorCategory": "数据集",
    "subcategories": [
      "机器人示范与操作数据",
      "跨机器人与多任务数据",
      "数据采集接口"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-9fc7047b7d787fa298bf",
    "title": "TD-MPC2: Scalable, Robust World Models for Continuous Control",
    "authors": "Nicklas Hansen; Hao Su; Xiaolong Wang",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "td_mpc2",
    "arxivUrl": null,
    "codeUrls": [
      "https://github.com/nicklashansen/tdmpc2"
    ],
    "projectUrl": "https://tdmpc2.com",
    "venue": "ICLR 2024",
    "paperUrl": "https://openreview.net/forum?id=Oxh5CstDJU",
    "pdfUrl": "https://arxiv.org/pdf/2310.16828",
    "doi": null,
    "publicationYear": 2024,
    "bibtex": "@inproceedings{td_mpc2,\n  author       = {Nicklas Hansen and\n                  Hao Su and\n                  Xiaolong Wang},\n  title        = {{TD-MPC2:} Scalable, Robust World Models for Continuous Control},\n  booktitle    = {ICLR},\n  year         = {2024},\n\n}",
    "majorCategory": "奠基性工作",
    "subcategories": [
      "经典WM与模型式RL",
      "理论与规划"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-9beb9ce37b602d31a5d8",
    "title": "SERL: A Software Suite for Sample-Efficient Robotic Reinforcement Learning",
    "authors": "Jianlan Luo; Zheyuan Hu; Charles Xu; You Liang Tan; Jacob Berg; Archit Sharma; Stefan Schaal; Chelsea Finn; Abhishek Gupta; Sergey Levine",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "serl",
    "arxivUrl": null,
    "codeUrls": [
      "https://github.com/rail-berkeley/serl"
    ],
    "projectUrl": "https://serl-robot.github.io/",
    "venue": "ICRA 2024",
    "paperUrl": "https://ieeexplore.ieee.org/document/10610040/",
    "pdfUrl": "https://arxiv.org/pdf/2401.16013",
    "doi": "https://doi.org/10.1109/ICRA57147.2024.10610040",
    "publicationYear": 2024,
    "bibtex": "@inproceedings{serl,\n  author       = {Jianlan Luo and\n                  Zheyuan Hu and\n                  Charles Xu and\n                  You Liang Tan and\n                  Jacob Berg and\n                  Archit Sharma and\n                  Stefan Schaal and\n                  Chelsea Finn and\n                  Abhishek Gupta and\n                  Sergey Levine},\n  title        = {{SERL:} {A} Software Suite for Sample-Efficient Robotic Reinforcement\n                  Learning},\n  booktitle    = {ICRA},\n  pages        = {16961--16969},\n  year         = {2024},\n}",
    "majorCategory": "奠基性工作",
    "subcategories": [
      "动作策略基础",
      "综述与技术资源"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-92e1d4830223bf075ed8",
    "title": "VidMan: Exploiting Implicit Dynamics from Video Diffusion Model for Effective Robot Manipulation",
    "authors": "Youpeng Wen; Junfan Lin; Yi Zhu; Jianhua Han; Hang Xu; Shen Zhao; Xiaodan Liang",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "vidman",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "NeurIPS 2024",
    "paperUrl": "https://proceedings.neurips.cc/paper_files/paper/2024/hash/481c70828a4ff20d31a646cc6cc95f3d-Abstract-Conference.html",
    "pdfUrl": "https://arxiv.org/pdf/2411.09153",
    "doi": "https://doi.org/10.52202/079017-1298",
    "publicationYear": 2024,
    "bibtex": "@inproceedings{vidman,\n  author       = {Youpeng Wen and\n                  Junfan Lin and\n                  Yi Zhu and\n                  Jianhua Han and\n                  Hang Xu and\n                  Shen Zhao and\n                  Xiaodan Liang},\n  title        = {VidMan: Exploiting Implicit Dynamics from Video Diffusion Model for\n                  Effective Robot Manipulation},\n  booktitle    = {NeurIPS},\n  year         = {2024},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "视觉规划与IDM",
      "高效推理与实时控制"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-8be34f0cf0d2ab79e166",
    "title": "Improved Distribution Matching Distillation for Fast Image Synthesis",
    "authors": "Tianwei Yin; Michaël Gharbi; Taesung Park; Richard Zhang; Eli Shechtman; Frédo Durand; William T. Freeman",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "dmd2",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "NeurIPS 2024",
    "paperUrl": "https://proceedings.neurips.cc/paper_files/paper/2024/hash/54dcf25318f9de5a7a01f0a4125c541e-Abstract-Conference.html",
    "pdfUrl": "https://proceedings.neurips.cc/paper_files/paper/2024/file/54dcf25318f9de5a7a01f0a4125c541e-Paper-Conference.pdf",
    "doi": "https://doi.org/10.52202/079017-1505",
    "publicationYear": 2024,
    "bibtex": "@article{dmd2,\n  title={Improved distribution matching distillation for fast image synthesis},\n  author={Yin, Tianwei and Gharbi, Micha{\\\"e}l and Park, Taesung and Zhang, Richard and Shechtman, Eli and Durand, Fredo and Freeman, William T},\n  journal={NeurIPS},\n  volume={37},\n  pages={47455--47487},\n  year={2024}\n}",
    "majorCategory": "奠基性工作",
    "subcategories": [
      "训练优化与蒸馏",
      "扩散与流匹配基础"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "综述明确"
  },
  {
    "id": "ref-8432892bf821930fe230",
    "title": "On-Policy Distillation of Language Models: Learning from Self-Generated Mistakes",
    "authors": "Rishabh Agarwal; Nino Vieillard; Yongchao Zhou; Piotr Stanczyk; Sabela Ramos Garea; Matthieu Geist; Olivier Bachem",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "OPD",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "ICLR 2024",
    "paperUrl": "https://proceedings.iclr.cc/paper_files/paper/2024/hash/5be69a584901a26c521c2b51e40a4c20-Abstract-Conference.html",
    "pdfUrl": "https://proceedings.iclr.cc/paper_files/paper/2024/file/5be69a584901a26c521c2b51e40a4c20-Paper-Conference.pdf",
    "doi": null,
    "publicationYear": 2024,
    "bibtex": "@inproceedings{OPD,\n  title={On-policy distillation of language models: Learning from self-generated mistakes},\n  author={Agarwal, Rishabh and Vieillard, Nino and Zhou, Yongchao and Stanczyk, Piotr and Ramos Garea, Sabela and Geist, Matthieu and Bachem, Olivier},\n  booktitle={ICLR},\n  volume={2024},\n  pages={21246--21263},\n  year={2024}\n}",
    "majorCategory": "奠基性工作",
    "subcategories": [
      "训练优化与蒸馏"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "综述明确"
  },
  {
    "id": "ref-80d04bd5d03ee6f72663",
    "title": "Revisiting Sparse Rewards for Goal-Reaching Reinforcement Learning",
    "authors": "Gautham Vasan; Yan Wang; Fahim Shahriar; James Bergstra; Martin Jägersand; A. Rupam Mahmood",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "vasan2024revisiting",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "Reinforcement Learning Journal",
    "paperUrl": "https://rlj.cs.umass.edu/2024/papers/Paper231.html",
    "pdfUrl": "https://rlj.cs.umass.edu/2024/papers/RLJ_RLC_2024_231.pdf",
    "doi": null,
    "publicationYear": 2024,
    "bibtex": "@article{vasan2024revisiting,\n  author       = {Gautham Vasan and\n                  Yan Wang and\n                  Fahim Shahriar and\n                  James Bergstra and\n                  Martin J{\\\"{a}}gersand and\n                  A. Rupam Mahmood},\n  title        = {Revisiting Sparse Rewards for Goal-Reaching Reinforcement Learning},\n  journal      = {{RLJ}},\n  volume       = {4},\n  pages        = {1841--1854},\n  year         = {2024},\n  \n}",
    "majorCategory": "奠基性工作",
    "subcategories": [
      "动作策略基础",
      "训练优化与蒸馏"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-7007d29ea25213a1f404",
    "title": "OccWorld: Learning a 3D Occupancy World Model for Autonomous Driving",
    "authors": "Wenzhao Zheng; Weiliang Chen; Yuanhui Huang; Borui Zhang; Yueqi Duan; Jiwen Lu",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "occworld",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "ECCV 2024",
    "paperUrl": "https://eccv.ecva.net/virtual/2024/poster/2474",
    "pdfUrl": "https://arxiv.org/pdf/2311.16038",
    "doi": null,
    "publicationYear": 2024,
    "bibtex": "@inproceedings{occworld,\n  author       = {Wenzhao Zheng and\n                  Weiliang Chen and\n                  Yuanhui Huang and\n                  Borui Zhang and\n                  Yueqi Duan and\n                  Jiwen Lu},\n  title        = {OccWorld: Learning a 3D Occupancy World Model for Autonomous Driving},\n  booktitle    = {ECCV},\n  pages        = {55--72},\n  year         = {2024},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "联合视频动作建模",
      "三维多视角建模",
      "自动驾驶"
    ],
    "architecture": "One Model",
    "predictionParadigm": "联合预测",
    "quadrant": "Q1 · One Model × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-685493265d49d3ee589a",
    "title": "RoboCasa: Large-Scale Simulation of Everyday Tasks for Generalist Robots",
    "authors": "Soroush Nasiriany; Abhiram Maddukuri; Lance Zhang; Adeet Parikh; Aaron Lo; Abhishek Joshi; Ajay Mandlekar; Yuke Zhu",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "robocasa",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": "https://robocasa.ai/",
    "venue": "RSS 2024",
    "paperUrl": "https://www.roboticsproceedings.org/rss20/p050.html",
    "pdfUrl": "https://www.roboticsproceedings.org/rss20/p050.pdf",
    "doi": null,
    "publicationYear": 2024,
    "bibtex": "@inproceedings{robocasa,\n  title={RoboCasa: Large-Scale Simulation of Everyday Tasks for Generalist Robots},\n  author={Soroush Nasiriany and Abhiram Maddukuri and Lance Zhang and Adeet Parikh and Aaron Lo and Abhishek Joshi and Ajay Mandlekar and Yuke Zhu},\n  booktitle={RSS},\n  year={2024}\n}",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "机器人操作基准",
      "物理仿真",
      "合成数据与数据生成"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-604316201f6ab37b2f9a",
    "title": "DriveDreamer: Towards Real-world-driven World Models for Autonomous Driving",
    "authors": "Xiaofeng Wang; Zheng Zhu; Guan Huang; Xinze Chen; Jiagang Zhu; Jiwen Lu",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "drivedreamer",
    "arxivUrl": null,
    "codeUrls": [
      "https://github.com/JeffWang987/DriveDreamer"
    ],
    "projectUrl": "https://drivedreamer.github.io/",
    "venue": "ECCV 2024",
    "paperUrl": "https://eccv.ecva.net/virtual/2024/poster/994",
    "pdfUrl": "https://www.ecva.net/papers/eccv_2024/papers_ECCV/papers/06416.pdf",
    "doi": null,
    "publicationYear": 2024,
    "bibtex": "@inproceedings{drivedreamer,\n  title={Drivedreamer: Towards real-world-drive world models for autonomous driving},\n  author={Wang, Xiaofeng and Zhu, Zheng and Huang, Guan and Chen, Xinze and Zhu, Jiagang and Lu, Jiwen},\n  booktitle={ECCV},\n  pages={55--72},\n  year={2024},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "自动驾驶",
      "联合视频动作建模"
    ],
    "architecture": "One Model",
    "predictionParadigm": "联合预测",
    "quadrant": "Q1 · One Model × 联合预测",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-5c86e0d92b645b2b135f",
    "title": "One-step Diffusion with Distribution Matching Distillation",
    "authors": "Tianwei Yin; Michaël Gharbi; Richard Zhang; Eli Shechtman; Frédo Durand; William T. Freeman; Taesung Park",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "dmd",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": "https://tianweiy.github.io/dmd/",
    "venue": "CVPR 2024",
    "paperUrl": "https://openaccess.thecvf.com/content/CVPR2024/html/Yin_One-step_Diffusion_with_Distribution_Matching_Distillation_CVPR_2024_paper.html",
    "pdfUrl": "https://openaccess.thecvf.com/content/CVPR2024/papers/Yin_One-step_Diffusion_with_Distribution_Matching_Distillation_CVPR_2024_paper.pdf",
    "doi": null,
    "publicationYear": 2024,
    "bibtex": "@inproceedings{dmd,\n  title={One-step diffusion with distribution matching distillation},\n  author={Yin, Tianwei and Gharbi, Micha{\\\"e}l and Zhang, Richard and Shechtman, Eli and Durand, Fredo and Freeman, William T and Park, Taesung},\n  booktitle={CVPR},\n  pages={6613--6623},\n  year={2024}\n}",
    "majorCategory": "奠基性工作",
    "subcategories": [
      "训练优化与蒸馏",
      "扩散与流匹配基础"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-4d8794e1eba7bfc5be75",
    "title": "Genie: Generative Interactive Environments",
    "authors": "Jake Bruce; Michael D Dennis; Ashley Edwards; Jack Parker-Holder; Yuge Shi; Edward Hughes; Matthew Lai; Aditi Mavalankar; Richie Steigerwald; Chris Apps; Yusuf Aytar; Sarah Maria Elisabeth Bechtle; Feryal Behbahani; Stephanie C.Y. Chan; Nicolas Heess; Lucy Gonzalez; Simon Osindero; Sherjil Ozair; Scott Reed; Jingwei Zhang; Konrad Zolna; Jeff Clune; Nando De Freitas; Satinder Singh; Tim Rocktäschel",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "genie",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "ICML 2024",
    "paperUrl": "https://proceedings.mlr.press/v235/bruce24a.html",
    "pdfUrl": "https://arxiv.org/pdf/2402.15391",
    "doi": null,
    "publicationYear": 2024,
    "bibtex": "@inproceedings{genie,\n  author       = {Jake Bruce and\n                  Michael D. Dennis and\n                  Ashley Edwards and\n                  Jack Parker{-}Holder and\n                  Yuge Shi and\n                  Edward Hughes and\n                  Matthew Lai and\n                  Aditi Mavalankar and\n                  Richie Steigerwald and\n                  Chris Apps and\n                  Yusuf Aytar and\n                  Sarah Bechtle and\n                  Feryal M. P. Behbahani and\n                  Stephanie C. Y. Chan and\n                  Nicolas Heess and\n                  Lucy Gonzalez and\n                  Simon Osindero and\n                  Sherjil Ozair and\n                  Scott E. Reed and\n                  Jingwei Zhang and\n                  Konrad Zolna and\n                  Jeff Clune and\n                  Nando de Freitas and\n                  Satinder Singh and\n                  Tim Rockt{\\\"{a}}schel},\n  title        = {Genie: Generative Interactive Environments},\n  booktitle    = {ICML},\n  pages        = {4603--4623},\n  year         = {2024},\n}",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "神经世界模拟器",
      "潜动作预训练"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-3b350556ce83b51f84c8",
    "title": "Diffusion for World Modeling: Visual Details Matter in Atari",
    "authors": "Eloi Alonso; Adam Jelley; Vincent Micheli; Anssi Kanervisto; Amos Storkey; Tim Pearce; François Fleuret",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "diamond",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": "https://diamond-wm.github.io",
    "venue": "NeurIPS 2024",
    "paperUrl": "https://proceedings.neurips.cc/paper_files/paper/2024/hash/6bdde0373d53d4a501249547084bed43-Abstract-Conference.html",
    "pdfUrl": "https://proceedings.neurips.cc/paper_files/paper/2024/file/6bdde0373d53d4a501249547084bed43-Paper-Conference.pdf",
    "doi": "https://doi.org/10.52202/079017-1873",
    "publicationYear": 2024,
    "bibtex": "@inproceedings{diamond,\n  author       = {Eloi Alonso and\n                  Adam Jelley and\n                  Vincent Micheli and\n                  Anssi Kanervisto and\n                  Amos J. Storkey and\n                  Tim Pearce and\n                  Fran{\\c{c}}ois Fleuret},\n  title        = {Diffusion for World Modeling: Visual Details Matter in Atari},\n  booktitle    = {NeurIPS},\n  year         = {2024},\n}",
    "majorCategory": "奠基性工作",
    "subcategories": [
      "经典WM与模型式RL"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-3145193f87c43012f17a",
    "title": "Craftax: A Lightning-Fast Benchmark for Open-Ended Reinforcement Learning",
    "authors": "Michael Matthews; Michael Beukman; Benjamin Ellis; Mikayel Samvelyan; Matthew Thomas Jackson; Samuel Coward; Jakob Nicolaus Foerster",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "craftx",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "ICML 2024",
    "paperUrl": "https://proceedings.mlr.press/v235/matthews24a.html",
    "pdfUrl": "https://raw.githubusercontent.com/mlresearch/v235/main/assets/matthews24a/matthews24a.pdf",
    "doi": null,
    "publicationYear": 2024,
    "bibtex": "@inproceedings{craftx,\n  author       = {Michael T. Matthews and\n                  Michael Beukman and\n                  Benjamin Ellis and\n                  Mikayel Samvelyan and\n                  Matthew Thomas Jackson and\n                  Samuel Coward and\n                  Jakob Nicolaus Foerster},\n  title        = {Craftax: {A} Lightning-Fast Benchmark for Open-Ended Reinforcement\n                  Learning},\n  booktitle    = {ICML},\n  pages        = {35104--35137},\n  year         = {2024},\n}",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "强化学习基准",
      "物理仿真"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "综述明确"
  },
  {
    "id": "ref-2e934302c61e88be910f",
    "title": "Video generation models as world simulators",
    "authors": "OpenAI",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "sora",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "OpenAI Technical Report",
    "paperUrl": "https://openai.com/index/video-generation-models-as-world-simulators/",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": 2024,
    "bibtex": "@article{sora,\n  title={Video Generation Models as World Simulators},\n  author={OpenAI},\n  journal={Technical Report},\n  year={2024},\n  url={https://openai.com/research/video-generation-models-as-world-simulators}\n}",
    "majorCategory": "WAM Components",
    "subcategories": [
      "Video & world prediction backbones",
      "视频生成Backbone"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "综述明确"
  },
  {
    "id": "ref-1ef0de9829c6511558f2",
    "title": "CoPeD-Advancing Multi-Robot Collaborative Perception: A Comprehensive Dataset in Real-World Environments",
    "authors": "Yang Zhou; Long Quang; Carlos Nieto-Granda; Giuseppe Loianno",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "Zhou_2024",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "IEEE Robotics and Automation Letters",
    "paperUrl": "https://ieeexplore.ieee.org/document/10540310/",
    "pdfUrl": "https://arxiv.org/pdf/2405.14731",
    "doi": "https://doi.org/10.1109/lra.2024.3406207",
    "publicationYear": 2024,
    "bibtex": "@article{Zhou_2024,\n  title     = {CoPeD-Advancing Multi-Robot Collaborative Perception: A Comprehensive Dataset in Real-World Environments},\n  volume    = {9},\n  issn      = {2377-3774},\n  number    = {7},\n  journal   = {RA-L},\n  author    = {Zhou, Yang and Quang, Long and Nieto-Granda, Carlos and Loianno, Giuseppe},\n  year      = {2024},\n  pages     = {6416–6423}\n}",
    "majorCategory": "数据集",
    "subcategories": [
      "多机器人协作感知数据",
      "多传感器与空间标注"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-1c048eabe2faa444f31b",
    "title": "Octo: An Open-Source Generalist Robot Policy",
    "authors": "Dibya Ghosh; Homer Rich Walke; Karl Pertsch; Kevin Black; Oier Mees; Sudeep Dasari; Joey Hejna; Tobias Kreiman; Charles Xu; Jianlan Luo; You Liang Tan; Lawrence Yunliang Chen; Quan Vuong; Ted Xiao; Pannag R. Sanketi; Dorsa Sadigh; Chelsea Finn; Sergey Levine",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "octo_2023",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": "https://octo-models.github.io",
    "venue": "RSS 2024",
    "paperUrl": "https://www.roboticsproceedings.org/rss20/p090.pdf",
    "pdfUrl": "https://www.roboticsproceedings.org/rss20/p090.pdf",
    "doi": null,
    "publicationYear": 2024,
    "bibtex": "@inproceedings{octo_2023,\n  author       = {Dibya Ghosh and\n                  Homer Rich Walke and\n                  Karl Pertsch and\n                  Kevin Black and\n                  Oier Mees and\n                  Sudeep Dasari and\n                  Joey Hejna and\n                  Tobias Kreiman and\n                  Charles Xu and\n                  Jianlan Luo and\n                  You Liang Tan and\n                  Lawrence Yunliang Chen and\n                  Quan Vuong and\n                  Ted Xiao and\n                  Pannag R. Sanketi and\n                  Dorsa Sadigh and\n                  Chelsea Finn and\n                  Sergey Levine},\n  title        = {Octo: An Open-Source Generalist Robot Policy},\n  booktitle    = {RSS},\n  year         = {2024},\n\n}",
    "majorCategory": "VLA",
    "subcategories": [
      "扩散与流匹配VLA"
    ],
    "architecture": "One Model",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-06ccc4edca82d6457b34",
    "title": "Vision-Language Foundation Models as Effective Robot Imitators",
    "authors": "Xinghang Li; Minghuan Liu; Hanbo Zhang; Cunjun Yu; Jie Xu; Hongtao Wu; Chilam Cheang; Ya Jing; Weinan Zhang; Huaping Liu; Hang Li; Tao Kong",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "li2023vision",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "ICLR 2024",
    "paperUrl": "https://openreview.net/forum?id=lFYj0oibGR",
    "pdfUrl": "https://openreview.net/pdf?id=lFYj0oibGR",
    "doi": null,
    "publicationYear": 2024,
    "bibtex": "@inproceedings{li2023vision,\n  author       = {Xinghang Li and\n                  Minghuan Liu and\n                  Hanbo Zhang and\n                  Cunjun Yu and\n                  Jie Xu and\n                  Hongtao Wu and\n                  Chilam Cheang and\n                  Ya Jing and\n                  Weinan Zhang and\n                  Huaping Liu and\n                  Hang Li and\n                  Tao Kong},\n  title        = {Vision-Language Foundation Models as Effective Robot Imitators},\n  booktitle    = {ICLR},\n  year         = {2024},\n\n}",
    "majorCategory": "VLA",
    "subcategories": [
      "分层与双系统VLA"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-02a3d941412699a67ab3",
    "title": "Prismatic VLMs: Investigating the Design Space of Visually-Conditioned Language Models",
    "authors": "Siddharth Karamcheti; Suraj Nair; Ashwin Balakrishna; Percy Liang; Thomas Kollar; Dorsa Sadigh",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "prismatic",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "ICML 2024",
    "paperUrl": "https://proceedings.mlr.press/v235/karamcheti24a.html",
    "pdfUrl": "https://raw.githubusercontent.com/mlresearch/v235/main/assets/karamcheti24a/karamcheti24a.pdf",
    "doi": null,
    "publicationYear": 2024,
    "bibtex": "@inproceedings{prismatic,\n  author       = {Siddharth Karamcheti and\n                  Suraj Nair and\n                  Ashwin Balakrishna and\n                  Percy Liang and\n                  Thomas Kollar and\n                  Dorsa Sadigh},\n  title        = {Prismatic VLMs: Investigating the Design Space of Visually-Conditioned\n                  Language Models},\n  booktitle    = {ICML},\n  pages        = {23123--23144},\n  year         = {2024},\n\n}",
    "majorCategory": "WAM Components",
    "subcategories": [
      "Language & vision-language backbones",
      "语言与VLM Backbone"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "综述明确"
  },
  {
    "id": "ref-01d0153fc9a4c1a98627",
    "title": "Open X-Embodiment: Robotic Learning Datasets and RT-X Models",
    "authors": "Open X-Embodiment Collaboration; Abby O'Neill; Abdul Rehman; Abhinav Gupta; Abhiram Maddukuri; Abhishek Gupta; Abhishek Padalkar; Abraham Lee; Acorn Pooley; Agrim Gupta; Ajay Mandlekar; Ajinkya Jain; Albert Tung; Alex Bewley; Alex Herzog; Alex Irpan; Alexander Khazatsky; Anant Rai; Anchit Gupta; Andrew Wang; Andrey Kolobov; Anikait Singh; Animesh Garg; Aniruddha Kembhavi; Annie Xie; Anthony Brohan; Antonin Raffin; Archit Sharma; Arefeh Yavary; Arhan Jain; Ashwin Balakrishna; Ayzaan Wahid; Ben Burgess-Limerick; Beomjoon Kim; Bernhard Schölkopf; Blake Wulfe; Brian Ichter; Cewu Lu; Charles Xu; Charlotte Le; Chelsea Finn; Chen Wang; Chenfeng Xu; Cheng Chi; Chenguang Huang; Christine Chan; Christopher Agia; Chuer Pan; Chuyuan Fu; Coline Devin; Danfei Xu; Daniel Morton; Danny Driess; Daphne Chen; Deepak Pathak; Dhruv Shah; Dieter Büchler; Dinesh Jayaraman; Dmitry Kalashnikov; Dorsa Sadigh; Edward Johns; Ethan Foster; Fangchen Liu; Federico Ceola; Fei Xia; Feiyu Zhao; Felipe Vieira Frujeri; Freek Stulp; Gaoyue Zhou; Gaurav S. Sukhatme; Gautam Salhotra; Ge Yan; Gilbert Feng; Giulio Schiavi; Glen Berseth; Gregory Kahn; Guangwen Yang; Guanzhi Wang; Hao Su; Hao-Shu Fang; Haochen Shi; Henghui Bao; Heni Ben Amor; Henrik I Christensen; Hiroki Furuta; Homanga Bharadhwaj; Homer Walke; Hongjie Fang; Huy Ha; Igor Mordatch; Ilija Radosavovic; Isabel Leal; Jacky Liang; Jad Abou-Chakra; Jaehyung Kim; Jaimyn Drake; Jan Peters; Jan Schneider; Jasmine Hsu; Jay Vakil; Jeannette Bohg; Jeffrey Bingham; Jeffrey Wu; Jensen Gao; Jiaheng Hu; Jiajun Wu; Jialin Wu; Jiankai Sun; Jianlan Luo; Jiayuan Gu; Jie Tan; Jihoon Oh; Jimmy Wu; Jingpei Lu; Jingyun Yang; Jitendra Malik; João Silvério; Joey Hejna; Jonathan Booher; Jonathan Tompson; Jonathan Yang; Jordi Salvador; Joseph J. Lim; Junhyek Han; Kaiyuan Wang; Kanishka Rao; Karl Pertsch; Karol Hausman; Keegan Go; Keerthana Gopalakrishnan; Ken Goldberg; Kendra Byrne; Kenneth Oslund; Kento Kawaharazuka; Kevin Black; Kevin Lin; Kevin Zhang; Kiana Ehsani; Kiran Lekkala; Kirsty Ellis; Krishan Rana; Krishnan Srinivasan; Kuan Fang; Kunal Pratap Singh; Kuo-Hao Zeng; Kyle Hatch; Kyle Hsu; Laurent Itti; Lawrence Yunliang Chen; Lerrel Pinto; Li Fei-Fei; Liam Tan; Linxi \"Jim\" Fan; Lionel Ott; Lisa Lee; Luca Weihs; Magnum Chen; Marion Lepert; Marius Memmel; Masayoshi Tomizuka; Masha Itkina; Mateo Guaman Castro; Max Spero; Maximilian Du; Michael Ahn; Michael C. Yip; Mingtong Zhang; Mingyu Ding; Minho Heo; Mohan Kumar Srirama; Mohit Sharma; Moo Jin Kim; Muhammad Zubair Irshad; Naoaki Kanazawa; Nicklas Hansen; Nicolas Heess; Nikhil J Joshi; Niko Suenderhauf; Ning Liu; Norman Di Palo; Nur Muhammad Mahi Shafiullah; Oier Mees; Oliver Kroemer; Osbert Bastani; Pannag R Sanketi; Patrick \"Tree\" Miller; Patrick Yin; Paul Wohlhart; Peng Xu; Peter David Fagan; Peter Mitrano; Pierre Sermanet; Pieter Abbeel; Priya Sundaresan; Qiuyu Chen; Quan Vuong; Rafael Rafailov; Ran Tian; Ria Doshi; Roberto Martín-Martín; Rohan Baijal; Rosario Scalise; Rose Hendrix; Roy Lin; Runjia Qian; Ruohan Zhang; Russell Mendonca; Rutav Shah; Ryan Hoque; Ryan Julian; Samuel Bustamante; Sean Kirmani; Sergey Levine; Shan Lin; Sherry Moore; Shikhar Bahl; Shivin Dass; Shubham Sonawani; Shubham Tulsiani; Shuran Song; Sichun Xu; Siddhant Haldar; Siddharth Karamcheti; Simeon Adebola; Simon Guist; Soroush Nasiriany; Stefan Schaal; Stefan Welker; Stephen Tian; Subramanian Ramamoorthy; Sudeep Dasari; Suneel Belkhale; Sungjae Park; Suraj Nair; Suvir Mirchandani; Takayuki Osa; Tanmay Gupta; Tatsuya Harada; Tatsuya Matsushima; Ted Xiao; Thomas Kollar; Tianhe Yu; Tianli Ding; Todor Davchev; Tony Z. Zhao; Travis Armstrong; Trevor Darrell; Trinity Chung; Vidhi Jain; Vikash Kumar; Vincent Vanhoucke; Vitor Guizilini; Wei Zhan; Wenxuan Zhou; Wolfram Burgard; Xi Chen; Xiangyu Chen; Xiaolong Wang; Xinghao Zhu; Xinyang Geng; Xiyuan Liu; Xu Liangwei; Xuanlin Li; Yansong Pang; Yao Lu; Yecheng Jason Ma; Yejin Kim; Yevgen Chebotar; Yifan Zhou; Yifeng Zhu; Yilin Wu; Ying Xu; Yixuan Wang; Yonatan Bisk; Yongqiang Dou; Yoonyoung Cho; Youngwoon Lee; Yuchen Cui; Yue Cao; Yueh-Hua Wu; Yujin Tang; Yuke Zhu; Yunchu Zhang; Yunfan Jiang; Yunshuang Li; Yunzhu Li; Yusuke Iwasawa; Yutaka Matsuo; Zehan Ma; Zhuo Xu; Zichen Jeff Cui; Zichen Zhang; Zipeng Fu; Zipeng Lin",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "oxe",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "2024 IEEE International Conference on Robotics and Automation (ICRA)",
    "paperUrl": "https://ieeexplore.ieee.org/document/10611477",
    "pdfUrl": "https://arxiv.org/pdf/2310.08864",
    "doi": "https://doi.org/10.1109/icra57147.2024.10611477",
    "publicationYear": 2024,
    "bibtex": "@inproceedings{oxe,\n  title={Open x-embodiment: Robotic learning datasets and rt-x models: Open x-embodiment collaboration 0},\n  author={O'Neill, Abby and Rehman, Abdul and Maddukuri, Abhiram and Gupta, Abhishek and Padalkar, Abhishek and Lee, Abraham and Pooley, Acorn and Gupta, Agrim and Mandlekar, Ajay and Jain, Ajinkya and others},\n  booktitle={ICRA},\n  pages={6892--6903},\n  year={2024},\n}",
    "majorCategory": "数据集",
    "subcategories": [
      "机器人示范与操作数据",
      "跨机器人与多任务数据",
      "语言标注与再标注"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2412.05337",
    "title": "ACT-Bench: Towards Action Controllable World Models for Autonomous Driving",
    "authors": "Hidehisa Arai; Keishi Ishihara; Tsubasa Takahashi; Yu Yamaguchi",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "act_bench",
    "arxivUrl": "https://arxiv.org/abs/2412.05337",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2412.05337",
    "pdfUrl": "https://arxiv.org/pdf/2412.05337",
    "doi": null,
    "publicationYear": 2024,
    "bibtex": "@inproceedings{\nact_bench,\ntitle={{ACT}-Bench: Towards Action Controllable World Models for Autonomous Driving},\nauthor={Hidehisa Arai and Keishi Ishihara and Tsubasa Takahashi and Yu Yamaguchi},\nbooktitle={ICLR Workshop},\nyear={2025},\n}",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "世界模型评测基准",
      "自动驾驶基准",
      "评测协议与诊断"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2311.15127",
    "title": "Stable Video Diffusion: Scaling Latent Video Diffusion Models to Large Datasets",
    "authors": "Andreas Blattmann; Tim Dockhorn; Sumith Kulal; Daniel Mendelevitch; Maciej Kilian; Dominik Lorenz; Yam Levi; Zion English; Vikram Voleti; Adam Letts; Varun Jampani; Robin Rombach",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2023-11-25",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "svd",
    "arxivUrl": "https://arxiv.org/abs/2311.15127",
    "codeUrls": [
      "https://github.com/Stability-AI/generative-models"
    ],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2311.15127",
    "pdfUrl": "https://arxiv.org/pdf/2311.15127",
    "doi": null,
    "publicationYear": 2023,
    "bibtex": "@article{svd,\n      title={Stable video diffusion: Scaling latent video diffusion models to large datasets}, \n      author={Blattmann, Andreas and Dockhorn, Tim and Kulal, Sumith and Mendelevitch, Daniel and Kilian, Maciej and Lorenz, Dominik and Levi, Yam and English, Zion and Voleti, Vikram and Letts, Adam and others},\n      year={2023},\n      journal={arXiv:2311.15127},\n}",
    "majorCategory": "WAM Components",
    "subcategories": [
      "Video & world prediction backbones",
      "视频生成Backbone"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2311.13549",
    "title": "ADriver-I: A General World Model for Autonomous Driving",
    "authors": "Fan Jia; Weixin Mao; Yingfei Liu; Yucheng Zhao; Yuqing Wen; Chi Zhang; Xiangyu Zhang; Tiancai Wang",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2023-11-22",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "adriver_i",
    "arxivUrl": "https://arxiv.org/abs/2311.13549",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2311.13549",
    "pdfUrl": "https://arxiv.org/pdf/2311.13549",
    "doi": null,
    "publicationYear": 2023,
    "bibtex": "@article{adriver_i,\n      title={ADriver-I: A General World Model for Autonomous Driving}, \n      author={Fan Jia and Weixin Mao and Yingfei Liu and Yucheng Zhao and Yuqing Wen and Chi Zhang and Xiangyu Zhang and Tiancai Wang},\n      year={2023},\n      journal={arXiv:2311.13549},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "联合视频动作建模",
      "自动驾驶"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2309.17080",
    "title": "GAIA-1: A Generative World Model for Autonomous Driving",
    "authors": "Anthony Hu; Lloyd Russell; Hudson Yeo; Zak Murez; George Fedoseev; Alex Kendall; Jamie Shotton; Gianluca Corrado",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2023-09-29",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "gaia_1",
    "arxivUrl": "https://arxiv.org/abs/2309.17080",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2309.17080",
    "pdfUrl": "https://arxiv.org/pdf/2309.17080",
    "doi": null,
    "publicationYear": 2023,
    "bibtex": "@article{gaia_1,\n      title={GAIA-1: A Generative World Model for Autonomous Driving}, \n      author={Anthony Hu and Lloyd Russell and Hudson Yeo and Zak Murez and George Fedoseev and Alex Kendall and Jamie Shotton and Gianluca Corrado},\n      year={2023},\n      journal={arXiv:2309.17080},\n}",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "神经世界模拟器"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2308.12966",
    "title": "Qwen-VL: A Versatile Vision-Language Model for Understanding, Localization, Text Reading, and Beyond",
    "authors": "Jinze Bai; Shuai Bai; Shusheng Yang; Shijie Wang; Sinan Tan; Peng Wang; Junyang Lin; Chang Zhou; Jingren Zhou",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2023-08-24",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "qwen-vl",
    "arxivUrl": "https://arxiv.org/abs/2308.12966",
    "codeUrls": [
      "https://github.com/QwenLM/Qwen-VL"
    ],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2308.12966",
    "pdfUrl": "https://arxiv.org/pdf/2308.12966",
    "doi": null,
    "publicationYear": 2023,
    "bibtex": "@article{qwen-vl,\n      title={Qwen-VL: A Versatile Vision-Language Model for Understanding, Localization, Text Reading, and Beyond}, \n      author={Jinze Bai and Shuai Bai and Shusheng Yang and Shijie Wang and Sinan Tan and Peng Wang and Junyang Lin and Chang Zhou and Jingren Zhou},\n      year={2023},\n      journal={arXiv:2308.12966},\n}",
    "majorCategory": "WAM Components",
    "subcategories": [
      "Language & vision-language backbones",
      "语言与VLM Backbone"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2302.13971",
    "title": "LLaMA: Open and Efficient Foundation Language Models",
    "authors": "Hugo Touvron; Thibaut Lavril; Gautier Izacard; Xavier Martinet; Marie-Anne Lachaux; Timothée Lacroix; Baptiste Rozière; Naman Goyal; Eric Hambro; Faisal Azhar; Aurelien Rodriguez; Armand Joulin; Edouard Grave; Guillaume Lample",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2023-02-27",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "llama",
    "arxivUrl": "https://arxiv.org/abs/2302.13971",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2302.13971",
    "pdfUrl": "https://arxiv.org/pdf/2302.13971",
    "doi": null,
    "publicationYear": 2023,
    "bibtex": "@article{llama,\n      title={LLaMA: Open and Efficient Foundation Language Models}, \n      author={Hugo Touvron and Thibaut Lavril and Gautier Izacard and Xavier Martinet and Marie-Anne Lachaux and Timothée Lacroix and Baptiste Rozière and Naman Goyal and Eric Hambro and Faisal Azhar and Aurelien Rodriguez and Armand Joulin and Edouard Grave and Guillaume Lample},\n      year={2023},\n      journal={arXiv:2302.13971},\n}",
    "majorCategory": "WAM Components",
    "subcategories": [
      "Language & vision-language backbones",
      "语言与VLM Backbone"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2302.00111",
    "title": "Learning Universal Policies via Text-Guided Video Generation",
    "authors": "Yilun Du; Mengjiao Yang; Bo Dai; Hanjun Dai; Ofir Nachum; Joshua B. Tenenbaum; Dale Schuurmans; Pieter Abbeel",
    "affiliations": "MIT; Google DeepMind; UC Berkeley; Georgia Tech; University of Alberta",
    "contribution": "UniPi turns a language instruction and current image into a video plan, refines its timing, then uses a separately trained inverse model to produce robot controls. Simulated manipulation results support compositional and multitask transfer. Internet pretraining improves generated real-scene plans, but its reported success is a classifier judgment on imagined final frames, not measured physical execution. [e02, e04, e06, e10, e14, e16]",
    "abstract": "",
    "submittedDate": "2023-01-31",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv230200111",
    "arxivUrl": "https://arxiv.org/abs/2302.00111",
    "codeUrls": [],
    "projectUrl": null,
    "venue": "NeurIPS 2023",
    "paperUrl": "https://arxiv.org/abs/2302.00111",
    "pdfUrl": "https://arxiv.org/pdf/2302.00111",
    "doi": null,
    "publicationYear": 2023,
    "bibtex": "",
    "majorCategory": "WAM",
    "subcategories": [
      "视觉规划与IDM"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "IDM",
    "quadrant": "Q4 · Dual-system × IDM",
    "classificationStatus": null
  },
  {
    "id": "ref-fccce46467ad720043ea",
    "title": "3D Gaussian Splatting for Real-Time Radiance Field Rendering",
    "authors": "Bernhard Kerbl; Georgios Kopanas; Thomas Leimkühler; George Drettakis",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "3dgs",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "ACM Transactions on Graphics 42(4), 2023",
    "paperUrl": "https://doi.org/10.1145/3592433",
    "pdfUrl": "https://arxiv.org/pdf/2308.04079",
    "doi": "https://doi.org/10.1145/3592433",
    "publicationYear": 2023,
    "bibtex": "@article{3dgs,\n  author       = {Bernhard Kerbl and\n                  Georgios Kopanas and\n                  Thomas Leimk{\\\"{u}}hler and\n                  George Drettakis},\n  title        = {3D Gaussian Splatting for Real-Time Radiance Field Rendering},\n  journal      = {ACM TOG},\n  volume       = {42},\n  number       = {4},\n  pages        = {139:1--139:14},\n  year         = {2023},\n\n}",
    "majorCategory": "WAM Components",
    "subcategories": [
      "Spatial perception & geometry",
      "三维表示与状态估计"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-daaf554d1668f36b7aa3",
    "title": "PaLM-E: An Embodied Multimodal Language Model",
    "authors": "Danny Driess; Fei Xia; Mehdi S. M. Sajjadi; Corey Lynch; Aakanksha Chowdhery; Brian Ichter; Ayzaan Wahid; Jonathan Tompson; Quan Vuong; Tianhe Yu; Wenlong Huang; Yevgen Chebotar; Pierre Sermanet; Daniel Duckworth; Sergey Levine; Vincent Vanhoucke; Karol Hausman; Marc Toussaint; Klaus Greff; Andy Zeng; Igor Mordatch; Pete Florence",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "palme",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": "https://palm-e.github.io/",
    "venue": "ICML 2023",
    "paperUrl": "https://proceedings.mlr.press/v202/driess23a.html",
    "pdfUrl": "https://arxiv.org/pdf/2303.03378",
    "doi": null,
    "publicationYear": 2023,
    "bibtex": "@inproceedings{palme,\n  author       = {Danny Driess and\n                  Fei Xia and\n                  Mehdi S. M. Sajjadi and\n                  Corey Lynch and\n                  Aakanksha Chowdhery and\n                  Brian Ichter and\n                  Ayzaan Wahid and\n                  Jonathan Tompson and\n                  Quan Vuong and\n                  Tianhe Yu and\n                  Wenlong Huang and\n                  Yevgen Chebotar and\n                  Pierre Sermanet and\n                  Daniel Duckworth and\n                  Sergey Levine and\n                  Vincent Vanhoucke and\n                  Karol Hausman and\n                  Marc Toussaint and\n                  Klaus Greff and\n                  Andy Zeng and\n                  Igor Mordatch and\n                  Pete Florence},\n  title        = {PaLM-E: An Embodied Multimodal Language Model},\n  booktitle    = {ICML},\n  pages        = {8469--8488},\n  year         = {2023},\n}",
    "majorCategory": "WAM Components",
    "subcategories": [
      "Language & vision-language backbones",
      "语言与VLM Backbone"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-d73223ab358ff6f8ecd9",
    "title": "RT-1: Robotics Transformer for Real-World Control at Scale",
    "authors": "Anthony Brohan; Noah Brown; Justice Carbajal; Yevgen Chebotar; Joseph Dabis; Chelsea Finn; Keerthana Gopalakrishnan; Karol Hausman; Alexander Herzog; Jasmine Hsu; Julian Ibarz; Brian Ichter; Alex Irpan; Tomas Jackson; Sally Jesmonth; Nikhil J. Joshi; Ryan Julian; Dmitry Kalashnikov; Yuheng Kuang; Isabel Leal; Kuang-Huei Lee; Sergey Levine; Yao Lu; Utsav Malla; Deeksha Manjunath; Igor Mordatch; Ofir Nachum; Carolina Parada; Jodilyn Peralta; Emily Perez; Karl Pertsch; Jornell Quiambao; Kanishka Rao; Michael S. Ryoo; Grecia Salazar; Pannag R. Sanketi; Kevin Sayed; Jaspiar Singh; Sumedh Sontakke; Austin Stone; Clayton Tan; Huong T. Tran; Vincent Vanhoucke; Steve Vega; Quan Vuong; Fei Xia; Ted Xiao; Peng Xu; Sichun Xu; Tianhe Yu; Brianna Zitkovich",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "rt1",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": "https://robotics-transformer1.github.io/",
    "venue": "RSS 2023",
    "paperUrl": "https://www.roboticsproceedings.org/rss19/p025.pdf",
    "pdfUrl": "https://www.roboticsproceedings.org/rss19/p025.pdf",
    "doi": null,
    "publicationYear": 2023,
    "bibtex": "@inproceedings{rt1,\n  author       = {Anthony Brohan and\n                  Noah Brown and\n                  Justice Carbajal and\n                  Yevgen Chebotar and\n                  Joseph Dabis and\n                  Chelsea Finn and\n                  Keerthana Gopalakrishnan and\n                  Karol Hausman and\n                  Alexander Herzog and\n                  Jasmine Hsu and\n                  Julian Ibarz and\n                  Brian Ichter and\n                  Alex Irpan and\n                  Tomas Jackson and\n                  Sally Jesmonth and\n                  Nikhil J. Joshi and\n                  Ryan Julian and\n                  Dmitry Kalashnikov and\n                  Yuheng Kuang and\n                  Isabel Leal and\n                  Kuang{-}Huei Lee and\n                  Sergey Levine and\n                  Yao Lu and\n                  Utsav Malla and\n                  Deeksha Manjunath and\n                  Igor Mordatch and\n                  Ofir Nachum and\n                  Carolina Parada and\n                  Jodilyn Peralta and\n                  Emily Perez and\n                  Karl Pertsch and\n                  Jornell Quiambao and\n                  Kanishka Rao and\n                  Michael S. Ryoo and\n                  Grecia Salazar and\n                  Pannag R. Sanketi and\n                  Kevin Sayed and\n                  Jaspiar Singh and\n                  Sumedh Sontakke and\n                  Austin Stone and\n                  Clayton Tan and\n                  Huong T. Tran and\n                  Vincent Vanhoucke and\n                  Steve Vega and\n                  Quan Vuong and\n                  Fei Xia and\n                  Ted Xiao and\n                  Peng Xu and\n                  Sichun Xu and\n                  Tianhe Yu and\n                  Brianna Zitkovich},\n  title        = {{RT-1:} Robotics Transformer for Real-World Control at Scale},\n  booktitle    = {RSS},\n  year         = {2023},\n\n}",
    "majorCategory": "VLA",
    "subcategories": [
      "离散动作VLA"
    ],
    "architecture": "One Model",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-d0d78a9a94fcee45c9aa",
    "title": "DexGraspNet: A Large-Scale Robotic Dexterous Grasp Dataset for General Objects Based on Simulation",
    "authors": "Ruicheng Wang; Jialiang Zhang; Jiayi Chen; Yinzhen Xu; Puhao Li; Tengyu Liu; He Wang",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "wang2023dexgraspnet",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": "https://pku-epic.github.io/DexGraspNet/",
    "venue": "ICRA 2023",
    "paperUrl": "https://ieeexplore.ieee.org/document/10160982/",
    "pdfUrl": "https://arxiv.org/pdf/2210.02697",
    "doi": "https://doi.org/10.1109/icra48891.2023.10160982",
    "publicationYear": 2023,
    "bibtex": "@inproceedings{wang2023dexgraspnet,\n  author       = {Ruicheng Wang and\n                  Jialiang Zhang and\n                  Jiayi Chen and\n                  Yinzhen Xu and\n                  Puhao Li and\n                  Tengyu Liu and\n                  He Wang},\n  title        = {DexGraspNet: {A} Large-Scale Robotic Dexterous Grasp Dataset for General\n                  Objects Based on Simulation},\n  booktitle    = {ICRA},\n  pages        = {11359--11366},\n  year         = {2023},\n\n}",
    "majorCategory": "数据集",
    "subcategories": [
      "灵巧手与抓取数据",
      "合成数据与数据生成",
      "三维物体资源"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-d0a7d699e0efc759ae64",
    "title": "BridgeData V2: A Dataset for Robot Learning at Scale",
    "authors": "Homer Walke; Kevin Black; Abraham Lee; Moo Jin Kim; Max Du; Chongyi Zheng; Tony Zhao; Philippe Hansen-Estruch; Quan Vuong; Andre He; Vivek Myers; Kuan Fang; Chelsea Finn; Sergey Levine",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "bridgev2",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "CoRL 2023",
    "paperUrl": "https://proceedings.mlr.press/v229/walke23a.html",
    "pdfUrl": "https://proceedings.mlr.press/v229/walke23a/walke23a.pdf",
    "doi": null,
    "publicationYear": 2023,
    "bibtex": "@inproceedings{bridgev2,\n  author       = {Homer Rich Walke and\n                  Kevin Black and\n                  Tony Z. Zhao and\n                  Quan Vuong and\n                  Chongyi Zheng and\n                  Philippe Hansen{-}Estruch and\n                  Andre Wang He and\n                  Vivek Myers and\n                  Moo Jin Kim and\n                  Max Du and\n                  Abraham Lee and\n                  Kuan Fang and\n                  Chelsea Finn and\n                  Sergey Levine},\n  title        = {BridgeData {V2:} {A} Dataset for Robot Learning at Scale},\n  booktitle    = {CoRL},\n  pages        = {1723--1736},\n  year         = {2023},\n\n}",
    "majorCategory": "数据集",
    "subcategories": [
      "机器人示范与操作数据",
      "跨机器人与多任务数据"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-d059266d6f51d5e33520",
    "title": "DREAMWALKER: Mental Planning for Continuous Vision-Language Navigation",
    "authors": "Hanqing Wang; Wei Liang; Luc Van Gool; Wenguan Wang",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "dreamwalker",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "ICCV 2023",
    "paperUrl": "https://openaccess.thecvf.com/content/ICCV2023/html/Wang_DREAMWALKER_Mental_Planning_for_Continuous_Vision-Language_Navigation_ICCV_2023_paper.html",
    "pdfUrl": "https://arxiv.org/pdf/2308.07498",
    "doi": null,
    "publicationYear": 2023,
    "bibtex": "@inproceedings{dreamwalker,\n  author       = {Hanqing Wang and\n                  Wei Liang and\n                  Luc Van Gool and\n                  Wenguan Wang},\n  title        = {Dreamwalker: Mental Planning for Continuous Vision-Language Navigation},\n  booktitle    = {ICCV},\n  pages        = {10839--10849},\n  year         = {2023},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "导航",
      "视觉规划与IDM"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-b4155f59c47b47bdc94d",
    "title": "Scalable Diffusion Models with Transformers",
    "authors": "William Peebles; Saining Xie",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "dit",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "ICCV 2023",
    "paperUrl": "https://openaccess.thecvf.com/content/ICCV2023/html/Peebles_Scalable_Diffusion_Models_with_Transformers_ICCV_2023_paper.html",
    "pdfUrl": "https://openaccess.thecvf.com/content/ICCV2023/papers/Peebles_Scalable_Diffusion_Models_with_Transformers_ICCV_2023_paper.pdf",
    "doi": null,
    "publicationYear": 2023,
    "bibtex": "@inproceedings{dit,\n  author       = {William Peebles and\n                  Saining Xie},\n  title        = {Scalable Diffusion Models with Transformers},\n  booktitle    = {ICCV},\n  pages        = {4172--4182},\n  year         = {2023},\n\n}",
    "majorCategory": "WAM Components",
    "subcategories": [
      "Generative modeling & tokenizers",
      "扩散与流匹配基础"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-8c87d43c8563d21b1a7e",
    "title": "Flow Straight and Fast: Learning to Generate and Transfer Data with Rectified Flow",
    "authors": "Xingchao Liu; Chengyue Gong; Qiang Liu",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "rectified_flow",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "ICLR 2023",
    "paperUrl": "https://openreview.net/forum?id=XVjTT1nw5z",
    "pdfUrl": "https://openreview.net/pdf?id=XVjTT1nw5z",
    "doi": null,
    "publicationYear": 2023,
    "bibtex": "@inproceedings{rectified_flow,\n  author       = {Xingchao Liu and\n                  Chengyue Gong and\n                  Qiang Liu},\n  title        = {Flow Straight and Fast: Learning to Generate and Transfer Data with\n                  Rectified Flow},\n  booktitle    = {ICLR},\n  year         = {2023},\n}",
    "majorCategory": "奠基性工作",
    "subcategories": [
      "扩散与流匹配基础"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "综述明确"
  },
  {
    "id": "ref-5677f2aa425315809584",
    "title": "MimicGen: A Data Generation System for Scalable Robot Learning using Human Demonstrations",
    "authors": "Ajay Mandlekar; Soroush Nasiriany; Bowen Wen; Iretiayo Akinola; Yashraj Narang; Linxi Fan; Yuke Zhu; Dieter Fox",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "mimicgen",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": "https://mimicgen.github.io",
    "venue": "CoRL 2023",
    "paperUrl": "https://proceedings.mlr.press/v229/mandlekar23a.html",
    "pdfUrl": "https://proceedings.mlr.press/v229/mandlekar23a/mandlekar23a.pdf",
    "doi": null,
    "publicationYear": 2023,
    "bibtex": "@inproceedings{mimicgen,\n  author       = {Ajay Mandlekar and\n                  Soroush Nasiriany and\n                  Bowen Wen and\n                  Iretiayo Akinola and\n                  Yashraj Narang and\n                  Linxi Fan and\n                  Yuke Zhu and\n                  Dieter Fox},\n  title        = {MimicGen: {A} Data Generation System for Scalable Robot Learning using\n                  Human Demonstrations},\n  booktitle    = {CoRL},\n  pages        = {1820--1864},\n  year         = {2023},\n}",
    "majorCategory": "数据集",
    "subcategories": [
      "合成数据与数据生成",
      "机器人示范与操作数据",
      "跨机器人与多任务数据"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-5514771916f15e54d7e6",
    "title": "DINO: DETR with Improved DeNoising Anchor Boxes for End-to-End Object Detection",
    "authors": "Hao Zhang; Feng Li; Shilong Liu; Lei Zhang; Hang Su; Jun Zhu; Lionel M. Ni; Heung-Yeung Shum",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "dino",
    "arxivUrl": null,
    "codeUrls": [
      "https://github.com/IDEA-Research/DINO"
    ],
    "projectUrl": null,
    "venue": "ICLR 2023",
    "paperUrl": "https://iclr.cc/virtual/2023/poster/11884",
    "pdfUrl": "https://arxiv.org/pdf/2203.03605",
    "doi": null,
    "publicationYear": 2023,
    "bibtex": "@inproceedings{dino,\n  author       = {Hao Zhang and\n                  Feng Li and\n                  Shilong Liu and\n                  Lei Zhang and\n                  Hang Su and\n                  Jun Zhu and\n                  Lionel M. Ni and\n                  Heung{-}Yeung Shum},\n  title        = {{DINO:} {DETR} with Improved DeNoising Anchor Boxes for End-to-End\n                  Object Detection},\n  booktitle    = {ICLR},\n  year         = {2023},\n \n}",
    "majorCategory": "Related resources",
    "subcategories": [
      "视觉编码器与表征"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-4fc7cd3abf32643da922",
    "title": "RT-2: Vision-Language-Action Models Transfer Web Knowledge to Robotic Control",
    "authors": "Brianna Zitkovich; Tianhe Yu; Sichun Xu; Peng Xu; Ted Xiao; Fei Xia; Jialin Wu; Paul Wohlhart; Stefan Welker; Ayzaan Wahid; Quan Vuong; Vincent Vanhoucke; Huong T. Tran; Radu Soricut; Anikait Singh; Jaspiar Singh; Pierre Sermanet; Pannag R. Sanketi; Grecia Salazar; Michael S. Ryoo; Krista Reymann; Kanishka Rao; Karl Pertsch; Igor Mordatch; Henryk Michalewski; Yao Lu; Sergey Levine; Lisa Lee; Tsang-Wei Edward Lee; Isabel Leal; Yuheng Kuang; Dmitry Kalashnikov; Ryan Julian; Nikhil J. Joshi; Alex Irpan; Brian Ichter; Jasmine Hsu; Alexander Herzog; Karol Hausman; Keerthana Gopalakrishnan; Chuyuan Fu; Pete Florence; Chelsea Finn; Kumar Avinava Dubey; Danny Driess; Tianli Ding; Krzysztof Marcin Choromanski; Xi Chen; Yevgen Chebotar; Justice Carbajal; Noah Brown; Anthony Brohan; Montserrat Gonzalez Arenas; Kehang Han",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "rt2",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "CoRL 2023",
    "paperUrl": "https://proceedings.mlr.press/v229/zitkovich23a.html",
    "pdfUrl": "https://arxiv.org/pdf/2307.15818",
    "doi": null,
    "publicationYear": 2023,
    "bibtex": "@inproceedings{rt2,\n  author       = {Brianna Zitkovich and\n                  Tianhe Yu and\n                  Sichun Xu and\n                  Peng Xu and\n                  Ted Xiao and\n                  Fei Xia and\n                  Jialin Wu and\n                  Paul Wohlhart and\n                  Stefan Welker and\n                  Ayzaan Wahid and\n                  Quan Vuong and\n                  Vincent Vanhoucke and\n                  Huong T. Tran and\n                  Radu Soricut and\n                  Anikait Singh and\n                  Jaspiar Singh and\n                  Pierre Sermanet and\n                  Pannag R. Sanketi and\n                  Grecia Salazar and\n                  Michael S. Ryoo and\n                  Krista Reymann and\n                  Kanishka Rao and\n                  Karl Pertsch and\n                  Igor Mordatch and\n                  Henryk Michalewski and\n                  Yao Lu and\n                  Sergey Levine and\n                  Lisa Lee and\n                  Tsang{-}Wei Edward Lee and\n                  Isabel Leal and\n                  Yuheng Kuang and\n                  Dmitry Kalashnikov and\n                  Ryan Julian and\n                  Nikhil J. Joshi and\n                  Alex Irpan and\n                  Brian Ichter and\n                  Jasmine Hsu and\n                  Alexander Herzog and\n                  Karol Hausman and\n                  Keerthana Gopalakrishnan and\n                  Chuyuan Fu and\n                  Pete Florence and\n                  Chelsea Finn and\n                  Kumar Avinava Dubey and\n                  Danny Driess and\n                  Tianli Ding and\n                  Krzysztof Marcin Choromanski and\n                  Xi Chen and\n                  Yevgen Chebotar and\n                  Justice Carbajal and\n                  Noah Brown and\n                  Anthony Brohan and\n                  Montserrat Gonzalez Arenas and\n                  Kehang Han},\n  title        = {{RT-2:} Vision-Language-Action Models Transfer Web Knowledge to Robotic\n                  Control},\n  booktitle    = {CoRL},\n  pages        = {2165--2183},\n  year         = {2023},\n\n}",
    "majorCategory": "VLA",
    "subcategories": [
      "自回归VLA"
    ],
    "architecture": "One Model",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-3fa175f81e25ece6b23f",
    "title": "Scaling Data Generation in Vision-and-Language Navigation",
    "authors": "Zun Wang; Jialu Li; Yicong Hong; Yi Wang; Qi Wu; Mohit Bansal; Stephen Gould; Hao Tan; Yu Qiao",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "scalevln",
    "arxivUrl": null,
    "codeUrls": [
      "https://github.com/wz0919/ScaleVLN"
    ],
    "projectUrl": null,
    "venue": "ICCV 2023",
    "paperUrl": "https://openaccess.thecvf.com/content/ICCV2023/papers/Wang_Scaling_Data_Generation_in_Vision-and-Language_Navigation_ICCV_2023_paper.pdf",
    "pdfUrl": "https://openaccess.thecvf.com/content/ICCV2023/papers/Wang_Scaling_Data_Generation_in_Vision-and-Language_Navigation_ICCV_2023_paper.pdf",
    "doi": null,
    "publicationYear": 2023,
    "bibtex": "@inproceedings{scalevln,\n  author       = {Zun Wang and\n                  Jialu Li and\n                  Yicong Hong and\n                  Yi Wang and\n                  Qi Wu and\n                  Mohit Bansal and\n                  Stephen Gould and\n                  Hao Tan and\n                  Yu Qiao},\n  title        = {Scaling Data Generation in Vision-and-Language Navigation},\n  booktitle    = {ICCV},\n  pages        = {11975--11986},\n  year         = {2023},\n}",
    "majorCategory": "数据集",
    "subcategories": [
      "合成数据与数据生成",
      "视觉语言导航数据",
      "语言标注与再标注"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-35f4994d647876b8cec0",
    "title": "DayDreamer: World Models for Physical Robot Learning",
    "authors": "Philipp Wu; Alejandro Escontrela; Danijar Hafner; Pieter Abbeel; Ken Goldberg",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "daydreamer",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "CoRL 2022",
    "paperUrl": "https://proceedings.mlr.press/v205/wu23c.html",
    "pdfUrl": "https://arxiv.org/pdf/2206.14176",
    "doi": null,
    "publicationYear": 2023,
    "bibtex": "@inproceedings{daydreamer,\n  author       = {Philipp Wu and\n                  Alejandro Escontrela and\n                  Danijar Hafner and\n                  Pieter Abbeel and\n                  Ken Goldberg},\n  title        = {DayDreamer: World Models for Physical Robot Learning},\n  booktitle    = {CoRL},\n  pages        = {2226--2240},\n  year         = {2022},\n}",
    "majorCategory": "奠基性工作",
    "subcategories": [
      "经典WM与模型式RL"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-355f71216b13cc8a4a3e",
    "title": "Transformers are Sample-Efficient World Models",
    "authors": "Vincent Micheli; Eloi Alonso; François Fleuret",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "iris",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "ICLR 2023",
    "paperUrl": "https://openreview.net/forum?id=vhFu1Acb0xb",
    "pdfUrl": "https://arxiv.org/pdf/2209.00588",
    "doi": null,
    "publicationYear": 2023,
    "bibtex": "@inproceedings{iris,\n  author       = {Vincent Micheli and\n                  Eloi Alonso and\n                  Fran{\\c{c}}ois Fleuret},\n  title        = {Transformers are Sample-Efficient World Models},\n  booktitle    = {ICLR},\n  year         = {2023},\n}",
    "majorCategory": "奠基性工作",
    "subcategories": [
      "经典WM与模型式RL"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-308fa479bc3588a7d13a",
    "title": "Sigmoid Loss for Language Image Pre-Training",
    "authors": "Xiaohua Zhai; Basil Mustafa; Alexander Kolesnikov; Lucas Beyer",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "siglip",
    "arxivUrl": null,
    "codeUrls": [
      "https://github.com/google-research/big_vision"
    ],
    "projectUrl": null,
    "venue": "ICCV 2023",
    "paperUrl": "https://openaccess.thecvf.com/content/ICCV2023/html/Zhai_Sigmoid_Loss_for_Language_Image_Pre-Training_ICCV_2023_paper.html",
    "pdfUrl": "https://arxiv.org/pdf/2303.15343",
    "doi": null,
    "publicationYear": 2023,
    "bibtex": "@inproceedings{siglip,\n  author       = {Xiaohua Zhai and\n                  Basil Mustafa and\n                  Alexander Kolesnikov and\n                  Lucas Beyer},\n  title        = {Sigmoid Loss for Language Image Pre-Training},\n  booktitle    = {ICCV},\n  pages        = {11941--11952},\n  year         = {2023},\n  \n}",
    "majorCategory": "WAM Components",
    "subcategories": [
      "视觉编码器与表征",
      "语言与VLM Backbone"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-3007c0da9fd6402981bf",
    "title": "LIBERO: Benchmarking Knowledge Transfer for Lifelong Robot Learning",
    "authors": "Bo Liu; Yifeng Zhu; Chongkai Gao; Yihao Feng; Qiang Liu; Yuke Zhu; Peter Stone",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "libero",
    "arxivUrl": null,
    "codeUrls": [
      "https://github.com/Lifelong-Robot-Learning/LIBERO"
    ],
    "projectUrl": "https://libero-project.github.io",
    "venue": "NeurIPS 2023 Datasets and Benchmarks Track",
    "paperUrl": "https://proceedings.nips.cc/paper_files/paper/2023/hash/8c3c666820ea055a77726d66fc7d447f-Abstract-Datasets_and_Benchmarks.html",
    "pdfUrl": "https://arxiv.org/pdf/2306.03310",
    "doi": "https://doi.org/10.52202/075280-1939",
    "publicationYear": 2023,
    "bibtex": "@inproceedings{libero,\n  author       = {Bo Liu and\n                  Yifeng Zhu and\n                  Chongkai Gao and\n                  Yihao Feng and\n                  Qiang Liu and\n                  Yuke Zhu and\n                  Peter Stone},\n  title        = {{LIBERO:} Benchmarking Knowledge Transfer for Lifelong Robot Learning},\n  booktitle    = {NeurIPS},\n  year         = {2023},\n}",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "机器人操作基准",
      "终身学习与知识迁移评测",
      "物理仿真"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-236eff7a0d0d26ae758e",
    "title": "OpenScene: The Largest Up-to-Date 3D Occupancy Prediction Benchmark in Autonomous Driving",
    "authors": "OpenScene Contributors",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "openscene",
    "arxivUrl": null,
    "codeUrls": [
      "https://github.com/OpenDriveLab/OpenScene"
    ],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://github.com/OpenDriveLab/OpenScene",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": 2023,
    "bibtex": "@misc{openscene,\n  author       = {{OpenScene Contributors}},\n  title        = {{OpenScene}: The Largest Up-to-Date 3D Occupancy Prediction Benchmark in Autonomous Driving},\n  howpublished = {\\url{https://github.com/OpenDriveLab/OpenScene}},\n  year         = {2023},\n}",
    "majorCategory": "数据集",
    "subcategories": [
      "自动驾驶感知数据",
      "三维占据与场景流数据",
      "基准与评测协议"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "综述明确"
  },
  {
    "id": "ref-12b97dfcece775bcc0b8",
    "title": "Zenseact Open Dataset: A large-scale and diverse multimodal dataset for autonomous driving",
    "authors": "Mina Alibeigi; William Ljungbergh; Adam Tonderski; Georg Hess; Adam Lilja; Carl Lindström; Daria Motorniuk; Junsheng Fu; Jenny Widahl; Christoffer Petersson",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "zod",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "ICCV 2023",
    "paperUrl": "https://openaccess.thecvf.com/content/ICCV2023/papers/Alibeigi_Zenseact_Open_Dataset_A_Large-Scale_and_Diverse_Multimodal_Dataset_for_ICCV_2023_paper.pdf",
    "pdfUrl": "https://openaccess.thecvf.com/content/ICCV2023/papers/Alibeigi_Zenseact_Open_Dataset_A_Large-Scale_and_Diverse_Multimodal_Dataset_for_ICCV_2023_paper.pdf",
    "doi": null,
    "publicationYear": 2023,
    "bibtex": "@inproceedings{zod,\n  author       = {Mina Alibeigi and\n                  William Ljungbergh and\n                  Adam Tonderski and\n                  Georg Hess and\n                  Adam Lilja and\n                  Carl Lindstr{\\\"{o}}m and\n                  Daria Motorniuk and\n                  Junsheng Fu and\n                  Jenny Widahl and\n                  Christoffer Petersson},\n  title        = {Zenseact Open Dataset: {A} large-scale and diverse multimodal dataset\n                  for autonomous driving},\n  booktitle    = {ICCV},\n  pages        = {20121--20131},\n  year         = {2023},\n}",
    "majorCategory": "数据集",
    "subcategories": [
      "自动驾驶感知数据",
      "运动预测与规划数据",
      "多传感器与空间标注"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-041a05059886890708fc",
    "title": "Consistency Models",
    "authors": "Yang Song; Prafulla Dhariwal; Mark Chen; Ilya Sutskever",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "cm",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "ICML 2023",
    "paperUrl": "https://proceedings.mlr.press/v202/song23a.html",
    "pdfUrl": "https://proceedings.mlr.press/v202/song23a/song23a.pdf",
    "doi": null,
    "publicationYear": 2023,
    "bibtex": "@InProceedings{cm,\n  title = \t {Consistency Models},\n  author =       {Song, Yang and Dhariwal, Prafulla and Chen, Mark and Sutskever, Ilya},\n  booktitle = \t {ICML},\n  pages = \t {32211--32252},\n  year = \t {2023},\n  volume = \t {202},\n  month = \t {23--29 Jul},\n}",
    "majorCategory": "奠基性工作",
    "subcategories": [
      "扩散与流匹配基础",
      "训练优化与蒸馏"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-f63a19593ae460d35e21",
    "title": "CALVIN: A Benchmark for Language-Conditioned Policy Learning for Long-Horizon Robot Manipulation Tasks",
    "authors": "Oier Mees; Lukás Hermann; Erick Rosete-Beas; Wolfram Burgard",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "calvin",
    "arxivUrl": null,
    "codeUrls": [
      "https://github.com/mees/calvin"
    ],
    "projectUrl": null,
    "venue": "IEEE Robotics and Automation Letters",
    "paperUrl": "https://ieeexplore.ieee.org/document/9788026",
    "pdfUrl": "https://arxiv.org/pdf/2112.03227",
    "doi": "https://doi.org/10.1109/LRA.2022.3180108",
    "publicationYear": 2022,
    "bibtex": "@article{calvin,\n  author       = {Oier Mees and\n                  Luk{\\'{a}}s Hermann and\n                  Erick Rosete{-}Beas and\n                  Wolfram Burgard},\n  title        = {{CALVIN:} {A} Benchmark for Language-Conditioned Policy Learning for\n                  Long-Horizon Robot Manipulation Tasks},\n  journal      = {RA-L},\n  volume       = {7},\n  number       = {3},\n  pages        = {7327--7334},\n  year         = {2022},\n}",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "机器人操作基准",
      "长时序与语言任务评测",
      "物理仿真"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-8a72842d518697e95077",
    "title": "Temporal Difference Learning for Model Predictive Control",
    "authors": "Nicklas A Hansen; Hao Su; Xiaolong Wang",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "td_mpc",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": "https://nicklashansen.github.io/td-mpc",
    "venue": "ICML 2022",
    "paperUrl": "https://proceedings.mlr.press/v162/hansen22a.html",
    "pdfUrl": "https://proceedings.mlr.press/v162/hansen22a/hansen22a.pdf",
    "doi": null,
    "publicationYear": 2022,
    "bibtex": "@inproceedings{td_mpc,\n  author       = {Nicklas Hansen and\n                  Hao Su and\n                  Xiaolong Wang},\n  title        = {Temporal Difference Learning for Model Predictive Control},\n  booktitle    = {ICML},\n  pages        = {8387--8406},\n  year         = {2022},\n\n}",
    "majorCategory": "奠基性工作",
    "subcategories": [
      "经典WM与模型式RL",
      "理论与规划"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-7874d48534f3bf22f87d",
    "title": "Rapid Exploration for Open-World Navigation with Latent Goal Models",
    "authors": "Dhruv Shah; Benjamin Eysenbach; Nicholas Rhinehart; Sergey Levine",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "recon",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": "https://sites.google.com/view/recon-robot",
    "venue": "CoRL 2021",
    "paperUrl": "https://proceedings.mlr.press/v164/shah22a.html",
    "pdfUrl": "https://proceedings.mlr.press/v164/shah22a/shah22a.pdf",
    "doi": null,
    "publicationYear": 2022,
    "bibtex": "@inproceedings{recon,\n  author       = {Dhruv Shah and\n                  Benjamin Eysenbach and\n                  Nicholas Rhinehart and\n                  Sergey Levine},\n  title        = {Rapid Exploration for Open-World Navigation with Latent Goal Models},\n  booktitle    = {CoRL},\n  pages        = {674--684},\n  year         = {2021},\n\n}",
    "majorCategory": "奠基性工作",
    "subcategories": [
      "动作策略基础",
      "理论与规划"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-71811bf04d6371d0bd82",
    "title": "LoRA: Low-Rank Adaptation of Large Language Models",
    "authors": "Edward J. Hu; Yelong Shen; Phillip Wallis; Zeyuan Allen-Zhu; Yuanzhi Li; Shean Wang; Lu Wang; Weizhu Chen",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "lora",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "ICLR 2022",
    "paperUrl": "https://openreview.net/forum?id=nZeVKeeFYf9",
    "pdfUrl": "https://openreview.net/pdf?id=nZeVKeeFYf9",
    "doi": null,
    "publicationYear": 2022,
    "bibtex": "@inproceedings{lora,\n  author       = {Edward J. Hu and\n                  Yelong Shen and\n                  Phillip Wallis and\n                  Zeyuan Allen{-}Zhu and\n                  Yuanzhi Li and\n                  Shean Wang and\n                  Lu Wang and\n                  Weizhu Chen},\n  title        = {LoRA: Low-Rank Adaptation of Large Language Models},\n  booktitle    = {ICLR},\n  year         = {2022},\n\n}",
    "majorCategory": "奠基性工作",
    "subcategories": [
      "训练优化与蒸馏"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-51f5ebd773eec33b2c6b",
    "title": "Ego4D: Around the World in 3,000 Hours of Egocentric Video",
    "authors": "Kristen Grauman; Andrew Westbury; Eugene Byrne; Zachary Chavis; Antonino Furnari; Rohit Girdhar; Jackson Hamburger; Hao Jiang; Miao Liu; Xingyu Liu; Miguel Martin; Tushar Nagarajan; Ilija Radosavovic; Santhosh Kumar Ramakrishnan; Fiona Ryan; Jayant Sharma; Michael Wray; Mengmeng Xu; Eric Zhongcong Xu; Chen Zhao; Siddhant Bansal; Dhruv Batra; Vincent Cartillier; Sean Crane; Tien Do; Morrie Doulaty; Akshay Erapalli; Christoph Feichtenhofer; Adriano Fragomeni; Qichen Fu; Abrham Gebreselasie; Cristina González; James Hillis; Xuhua Huang; Yifei Huang; Wenqi Jia; Weslie Khoo; Jáchym Kolár; Satwik Kottur; Anurag Kumar; Federico Landini; Chao Li; Yanghao Li; Zhenqiang Li; Karttikeya Mangalam; Raghava Modhugu; Jonathan Munro; Tullie Murrell; Takumi Nishiyasu; Will Price; Paola Ruiz Puentes; Merey Ramazanova; Leda Sari; Kiran K. Somasundaram; Audrey Southerland; Yusuke Sugano; Ruijie Tao; Minh Vo; Yuchen Wang; Xindi Wu; Takuma Yagi; Ziwei Zhao; Yunyi Zhu; Pablo Arbeláez; David Crandall; Dima Damen; Giovanni Maria Farinella; Christian Fuegen; Bernard Ghanem; Vamsi Krishna Ithapu; C. V. Jawahar; Hanbyul Joo; Kris Kitani; Haizhou Li; Richard A. Newcombe; Aude Oliva; Hyun Soo Park; James M. Rehg; Yoichi Sato; Jianbo Shi; Mike Zheng Shou; Antonio Torralba; Lorenzo Torresani; Mingfei Yan; Jitendra Malik",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "ego4d",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "CVPR 2022",
    "paperUrl": "https://openaccess.thecvf.com/content/CVPR2022/html/Grauman_Ego4D_Around_the_World_in_3000_Hours_of_Egocentric_Video_CVPR_2022_paper.html",
    "pdfUrl": "https://openaccess.thecvf.com/content/CVPR2022/papers/Grauman_Ego4D_Around_the_World_in_3000_Hours_of_Egocentric_Video_CVPR_2022_paper.pdf",
    "doi": null,
    "publicationYear": 2022,
    "bibtex": "@inproceedings{ego4d,\n  author       = {Kristen Grauman and\n                  Andrew Westbury and\n                  Eugene Byrne and\n                  Zachary Chavis and\n                  Antonino Furnari and\n                  Rohit Girdhar and\n                  Jackson Hamburger and\n                  Hao Jiang and\n                  Miao Liu and\n                  Xingyu Liu and\n                  Miguel Martin and\n                  Tushar Nagarajan and\n                  Ilija Radosavovic and\n                  Santhosh Kumar Ramakrishnan and\n                  Fiona Ryan and\n                  Jayant Sharma and\n                  Michael Wray and\n                  Mengmeng Xu and\n                  Eric Zhongcong Xu and\n                  Chen Zhao and\n                  Siddhant Bansal and\n                  Dhruv Batra and\n                  Vincent Cartillier and\n                  Sean Crane and\n                  Tien Do and\n                  Morrie Doulaty and\n                  Akshay Erapalli and\n                  Christoph Feichtenhofer and\n                  Adriano Fragomeni and\n                  Qichen Fu and\n                  Abrham Gebreselasie and\n                  Cristina Gonz{\\'{a}}lez and\n                  James Hillis and\n                  Xuhua Huang and\n                  Yifei Huang and\n                  Wenqi Jia and\n                  Weslie Khoo and\n                  J{\\'{a}}chym Kol{\\'{a}}r and\n                  Satwik Kottur and\n                  Anurag Kumar and\n                  Federico Landini and\n                  Chao Li and\n                  Yanghao Li and\n                  Zhenqiang Li and\n                  Karttikeya Mangalam and\n                  Raghava Modhugu and\n                  Jonathan Munro and\n                  Tullie Murrell and\n                  Takumi Nishiyasu and\n                  Will Price and\n                  Paola Ruiz Puentes and\n                  Merey Ramazanova and\n                  Leda Sari and\n                  Kiran K. Somasundaram and\n                  Audrey Southerland and\n                  Yusuke Sugano and\n                  Ruijie Tao and\n                  Minh Vo and\n                  Yuchen Wang and\n                  Xindi Wu and\n                  Takuma Yagi and\n                  Ziwei Zhao and\n                  Yunyi Zhu and\n                  Pablo Arbel{\\'{a}}ez and\n                  David Crandall and\n                  Dima Damen and\n                  Giovanni Maria Farinella and\n                  Christian Fuegen and\n                  Bernard Ghanem and\n                  Vamsi Krishna Ithapu and\n                  C. V. Jawahar and\n                  Hanbyul Joo and\n                  Kris Kitani and\n                  Haizhou Li and\n                  Richard A. Newcombe and\n                  Aude Oliva and\n                  Hyun Soo Park and\n                  James M. Rehg and\n                  Yoichi Sato and\n                  Jianbo Shi and\n                  Mike Zheng Shou and\n                  Antonio Torralba and\n                  Lorenzo Torresani and\n                  Mingfei Yan and\n                  Jitendra Malik},\n  title        = {Ego4D: Around the World in 3, 000 Hours of Egocentric Video},\n  booktitle    = {CVPR},\n  pages        = {18973--18990},\n  year         = {2022},\n\n}",
    "majorCategory": "数据集",
    "subcategories": [
      "人类第一视角数据",
      "多传感器与空间标注",
      "基准与评测协议"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-4db46d89f6231c67051e",
    "title": "Contrastive Learning as Goal-Conditioned Reinforcement Learning",
    "authors": "Benjamin Eysenbach; Tianjun Zhang; Sergey Levine; Russ R. Salakhutdinov",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "eysenbach2022contrastive",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "NeurIPS 2022",
    "paperUrl": "https://proceedings.neurips.cc/paper_files/paper/2022/hash/e7663e974c4ee7a2b475a4775201ce1f-Abstract-Conference.html",
    "pdfUrl": "https://proceedings.neurips.cc/paper/2022/file/e7663e974c4ee7a2b475a4775201ce1f-Paper-Conference.pdf",
    "doi": "https://doi.org/10.52202/068431-2580",
    "publicationYear": 2022,
    "bibtex": "@article{eysenbach2022contrastive,\n  title={Contrastive learning as goal-conditioned reinforcement learning},\n  author={Eysenbach, Benjamin and Zhang, Tianjun and Levine, Sergey and Salakhutdinov, Russ R},\n  journal={Advances in Neural Information Processing Systems},\n  volume={35},\n  pages={35603--35620},\n  year={2022}\n}",
    "majorCategory": "奠基性工作",
    "subcategories": [
      "理论与规划",
      "动作策略基础"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "综述明确"
  },
  {
    "id": "ref-3977414cc83281ab2172",
    "title": "Socially Compliant Navigation Dataset (SCAND): A Large-Scale Dataset of Demonstrations for Social Navigation",
    "authors": "Haresh Karnan; Anirudh Nair; Xuesu Xiao; Garrett Warnell; Sören Pirk; Alexander Toshev; Justin W. Hart; Joydeep Biswas; Peter Stone",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "scand",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "IEEE Robotics and Automation Letters",
    "paperUrl": "https://ieeexplore.ieee.org/document/9799755/",
    "pdfUrl": "https://arxiv.org/pdf/2203.15041",
    "doi": "https://doi.org/10.1109/lra.2022.3184025",
    "publicationYear": 2022,
    "bibtex": "@article{scand,\n  author       = {Haresh Karnan and\n                  Anirudh Nair and\n                  Xuesu Xiao and\n                  Garrett Warnell and\n                  S{\\\"{o}}ren Pirk and\n                  Alexander Toshev and\n                  Justin W. Hart and\n                  Joydeep Biswas and\n                  Peter Stone},\n  title        = {Socially CompliAnt Navigation Dataset {(SCAND):} {A} Large-Scale Dataset\n                  of Demonstrations for Social Navigation},\n  journal      = {RA-L},\n  volume       = {7},\n  number       = {4},\n  pages        = {11807--11814},\n  year         = {2022},\n\n}",
    "majorCategory": "数据集",
    "subcategories": [
      "导航与驾驶示范数据",
      "多传感器与空间标注"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-07638f74c59962f3e906",
    "title": "High-Resolution Image Synthesis With Latent Diffusion Models",
    "authors": "Robin Rombach; Andreas Blattmann; Dominik Lorenz; Patrick Esser; Björn Ommer",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "ldm",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "CVPR 2022",
    "paperUrl": "https://openaccess.thecvf.com/content/CVPR2022/html/Rombach_High-Resolution_Image_Synthesis_With_Latent_Diffusion_Models_CVPR_2022_paper.html",
    "pdfUrl": "https://arxiv.org/pdf/2112.10752",
    "doi": null,
    "publicationYear": 2022,
    "bibtex": "@inproceedings{ldm,\n  author       = {Robin Rombach and\n                  Andreas Blattmann and\n                  Dominik Lorenz and\n                  Patrick Esser and\n                  Bj{\\\"{o}}rn Ommer},\n  title        = {High-Resolution Image Synthesis with Latent Diffusion Models},\n  booktitle    = {CVPR},\n  pages        = {10674--10685},\n  year         = {2022},\n}",
    "majorCategory": "WAM Components",
    "subcategories": [
      "Generative modeling & tokenizers",
      "扩散与流匹配基础"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2205.01791",
    "title": "TartanDrive: A Large-Scale Dataset for Learning Off-Road Dynamics Models",
    "authors": "Samuel Triest; Matthew Sivaprakasam; Sean J. Wang; Wenshan Wang; Aaron M. Johnson; Sebastian A. Scherer",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "tartandrive",
    "arxivUrl": "https://arxiv.org/abs/2205.01791",
    "codeUrls": [
      "https://github.com/castacks/tartan_drive"
    ],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2205.01791",
    "pdfUrl": "https://arxiv.org/pdf/2205.01791",
    "doi": null,
    "publicationYear": 2022,
    "bibtex": "@inproceedings{tartandrive,\n  author       = {Samuel Triest and\n                  Matthew Sivaprakasam and\n                  Sean J. Wang and\n                  Wenshan Wang and\n                  Aaron M. Johnson and\n                  Sebastian A. Scherer},\n  title        = {TartanDrive: {A} Large-Scale Dataset for Learning Off-Road Dynamics\n                  Models},\n  booktitle    = {ICRA},\n  pages        = {2546--2552},\n  year         = {2022},\n\n}",
    "majorCategory": "数据集",
    "subcategories": [
      "导航与驾驶示范数据",
      "多传感器与空间标注",
      "运动预测与规划数据"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2112.11790",
    "title": "BEVDet: High-performance Multi-camera 3D Object Detection in Bird-Eye-View",
    "authors": "Junjie Huang; Guan Huang; Zheng Zhu; Yun Ye; Dalong Du",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2021-12-22",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "huang2023bevdet",
    "arxivUrl": "https://arxiv.org/abs/2112.11790",
    "codeUrls": [
      "https://github.com/HuangJunJie2017/BEVDet"
    ],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2112.11790",
    "pdfUrl": "https://arxiv.org/pdf/2112.11790",
    "doi": null,
    "publicationYear": 2021,
    "bibtex": "@article{huang2023bevdet,\n      title={BEVDet: High-performance Multi-camera 3D Object Detection in Bird-Eye-View}, \n      author={Junjie Huang and Guan Huang and Zheng Zhu and Yun Ye and Dalong Du},\n      year={2021},\n      journal={arXiv:2112.11790},\n}",
    "majorCategory": "WAM Components",
    "subcategories": [
      "Spatial perception & geometry",
      "三维表示与状态估计",
      "视觉编码器与表征"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2103.11943",
    "title": "BERT: A Review of Applications in Natural Language Processing and Understanding",
    "authors": "M. V. Koroteev",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2021-03-22",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "bert",
    "arxivUrl": "https://arxiv.org/abs/2103.11943",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2103.11943",
    "pdfUrl": "https://arxiv.org/pdf/2103.11943",
    "doi": null,
    "publicationYear": 2021,
    "bibtex": "@article{bert,\n      title={BERT: A Review of Applications in Natural Language Processing and Understanding}, \n      author={M. V. Koroteev},\n      year={2021},\n      journal={arXiv:2103.11943},\n}",
    "majorCategory": null,
    "subcategories": [
      "综述与技术资源"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-e1815cc67e6f9fc2bb7e",
    "title": "Mastering Atari with Discrete World Models",
    "authors": "Danijar Hafner; Timothy Lillicrap; Mohammad Norouzi; Jimmy Ba",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "dreamerv2",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "ICLR 2021",
    "paperUrl": "https://openreview.net/forum?id=0oabwyZbOu",
    "pdfUrl": "https://openreview.net/pdf?id=0oabwyZbOu",
    "doi": null,
    "publicationYear": 2021,
    "bibtex": "@inproceedings{dreamerv2,\n  author       = {Danijar Hafner and\n                  Timothy P. Lillicrap and\n                  Mohammad Norouzi and\n                  Jimmy Ba},\n  title        = {Mastering Atari with Discrete World Models},\n  booktitle    = {ICLR},\n  year         = {2021},\n \n}",
    "majorCategory": "奠基性工作",
    "subcategories": [
      "经典WM与模型式RL"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-dbc59b0e47889d094c8f",
    "title": "Pathdreamer: A World Model for Indoor Navigation",
    "authors": "Jing Yu Koh; Honglak Lee; Yinfei Yang; Jason Baldridge; Peter Anderson",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "pathdreamer",
    "arxivUrl": null,
    "codeUrls": [
      "https://github.com/google-research/pathdreamer"
    ],
    "projectUrl": "https://google-research.github.io/pathdreamer/",
    "venue": "ICCV 2021",
    "paperUrl": "https://openaccess.thecvf.com/content/ICCV2021/html/Koh_Pathdreamer_A_World_Model_for_Indoor_Navigation_ICCV_2021_paper.html",
    "pdfUrl": "https://arxiv.org/pdf/2105.08756",
    "doi": null,
    "publicationYear": 2021,
    "bibtex": "@inproceedings{pathdreamer,\n  author       = {Jing Yu Koh and\n                  Honglak Lee and\n                  Yinfei Yang and\n                  Jason Baldridge and\n                  Peter Anderson},\n  title        = {Pathdreamer: {A} World Model for Indoor Navigation},\n  booktitle    = {ICCV},\n  pages        = {14718--14728},\n  year         = {2021},\n}",
    "majorCategory": "WAM",
    "subcategories": [
      "导航",
      "记忆与长时序"
    ],
    "architecture": "Dual-system",
    "predictionParadigm": "其他机制",
    "quadrant": "四象限外",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-92f666cd9d15e83951cb",
    "title": "Deep Reinforcement Learning at the Edge of the Statistical Precipice",
    "authors": "Rishabh Agarwal; Max Schwarzer; Pablo Samuel Castro; Aaron Courville; Marc Bellemare",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "rliable",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "NeurIPS 2021",
    "paperUrl": "https://papers.neurips.cc/paper_files/paper/2021/hash/f514cec81cb148559cf475e7426eed5e-Abstract.html",
    "pdfUrl": "https://proceedings.neurips.cc/paper_files/paper/2021/file/f514cec81cb148559cf475e7426eed5e-Paper.pdf",
    "doi": null,
    "publicationYear": 2021,
    "bibtex": "@inproceedings{rliable,\n  title={Deep Reinforcement Learning at the Edge of the Statistical Precipice},\n  author={Agarwal, Rishabh and Schwarzer, Max and Castro, Pablo Samuel and Courville, Aaron and Bellemare, Marc G.},\n  booktitle={NeurIPS},\n  year={2021}\n}",
    "majorCategory": "评估指标（Metrics）",
    "subcategories": [
      "统计评测协议",
      "评估指标与协议"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-6f620ba80d9567d982a6",
    "title": "Learning Transferable Visual Models From Natural Language Supervision",
    "authors": "Alec Radford; Jong Wook Kim; Chris Hallacy; Aditya Ramesh; Gabriel Goh; Sandhini Agarwal; Girish Sastry; Amanda Askell; Pamela Mishkin; Jack Clark; Gretchen Krueger; Ilya Sutskever",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "clip",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "ICML 2021",
    "paperUrl": "https://proceedings.mlr.press/v139/radford21a.html",
    "pdfUrl": "https://proceedings.mlr.press/v139/radford21a/radford21a.pdf",
    "doi": null,
    "publicationYear": 2021,
    "bibtex": "@inproceedings{clip,\n  author       = {Alec Radford and\n                  Jong Wook Kim and\n                  Chris Hallacy and\n                  Aditya Ramesh and\n                  Gabriel Goh and\n                  Sandhini Agarwal and\n                  Girish Sastry and\n                  Amanda Askell and\n                  Pamela Mishkin and\n                  Jack Clark and\n                  Gretchen Krueger and\n                  Ilya Sutskever},\n  title        = {Learning Transferable Visual Models From Natural Language Supervision},\n  booktitle    = {ICML},\n  pages        = {8748--8763},\n  year         = {2021},\n}",
    "majorCategory": "WAM Components",
    "subcategories": [
      "视觉编码器与表征",
      "语言与VLM Backbone"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "综述明确"
  },
  {
    "id": "ref-18fe7d2bfb0e93f4c45e",
    "title": "ManiSkill: Generalizable Manipulation Skill Benchmark with Large-Scale Demonstrations",
    "authors": "Tongzhou Mu; Zhan Ling; Fanbo Xiang; Derek Yang; Xuanlin Li; Stone Tao; Zhiao Huang; Zhiwei Jia; Hao Su",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "maniskill",
    "arxivUrl": null,
    "codeUrls": [
      "https://github.com/haosulab/ManiSkill"
    ],
    "projectUrl": null,
    "venue": "NeurIPS 2021 Datasets and Benchmarks Track",
    "paperUrl": "https://datasets-benchmarks-proceedings.neurips.cc/paper_files/paper/2021/hash/eda80a3d5b344bc40f3bc04f65b7a357-Abstract-round2.html",
    "pdfUrl": "https://datasets-benchmarks-proceedings.neurips.cc/paper/2021/file/eda80a3d5b344bc40f3bc04f65b7a357-Paper-round2.pdf",
    "doi": null,
    "publicationYear": 2021,
    "bibtex": "@inproceedings{maniskill,\n  author       = {Tongzhou Mu and\n                  Zhan Ling and\n                  Fanbo Xiang and\n                  Derek Yang and\n                  Xuanlin Li and\n                  Stone Tao and\n                  Zhiao Huang and\n                  Zhiwei Jia and\n                  Hao Su},\n  title        = {ManiSkill: Generalizable Manipulation Skill Benchmark with Large-Scale\n                  Demonstrations},\n  booktitle    = {NeurIPS Datasets and Benchmarks},\n  year         = {2021},\n}",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "机器人操作基准",
      "物理仿真",
      "泛化评测",
      "示范数据"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "2106.11810",
    "title": "NuPlan: A closed-loop ML-based planning benchmark for autonomous vehicles",
    "authors": "Holger Caesar; Juraj Kabzan; Kok Seang Tan; Whye Kit Fong; Eric Wolff; Alex Lang; Luke Fletcher; Oscar Beijbom; Sammy Omari",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "nuplan",
    "arxivUrl": "https://arxiv.org/abs/2106.11810",
    "codeUrls": [],
    "projectUrl": "https://motional.com/news/nuplan-closed-loop-ml-based-planning-benchmark-autonomous-vehicles",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/2106.11810",
    "pdfUrl": "https://arxiv.org/pdf/2106.11810",
    "doi": null,
    "publicationYear": 2021,
    "bibtex": "@INPROCEEDINGS{nuplan, \n  title={NuPlan: A closed-loop ML-based planning benchmark for autonomous vehicles},\n  author={H. Caesar, J. Kabzan, K. Tan et al.},\n  booktitle={CVPR ADP3 workshop},\n  year=2021\n}",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "自动驾驶基准",
      "驾驶仿真",
      "驾驶数据",
      "评测协议与诊断"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-f851fa79e6baee5a919e",
    "title": "Beyond the Nav-Graph: Vision-and-Language Navigation in Continuous Environments",
    "authors": "Jacob Krantz; Erik Wijmans; Arjun Majumdar; Dhruv Batra; Stefan Lee",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "r2r_ce",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "ECCV 2020",
    "paperUrl": "https://www.ecva.net/papers/eccv_2020/papers_ECCV/html/6018_ECCV_2020_paper.php",
    "pdfUrl": "https://www.ecva.net/papers/eccv_2020/papers_ECCV/papers/123730103.pdf",
    "doi": null,
    "publicationYear": 2020,
    "bibtex": "@inproceedings{r2r_ce,\n  author       = {Jacob Krantz and\n                  Erik Wijmans and\n                  Arjun Majumdar and\n                  Dhruv Batra and\n                  Stefan Lee},\n  title        = {Beyond the Nav-Graph: Vision-and-Language Navigation in Continuous\n                  Environments},\n  booktitle    = {ECCV},\n  pages        = {104--120},\n  year         = {2020},\n}",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "导航基准",
      "物理仿真",
      "评测协议与诊断"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-b72f7e6ba4baaa80c600",
    "title": "Model-Based Reinforcement Learning for Atari",
    "authors": "Lukasz Kaiser; Mohammad Babaeizadeh; Piotr Milos; Blazej Osinski; Roy H. Campbell; Konrad Czechowski; Dumitru Erhan; Chelsea Finn; Piotr Kozakowski; Sergey Levine; Afroz Mohiuddin; Ryan Sepassi; George Tucker; Henryk Michalewski",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "atari100k",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "ICLR 2020",
    "paperUrl": "https://iclr.cc/virtual/2020/poster/1766",
    "pdfUrl": "https://arxiv.org/pdf/1903.00374",
    "doi": null,
    "publicationYear": 2020,
    "bibtex": "@inproceedings{atari100k,\n  author       = {Lukasz Kaiser and\n                  Mohammad Babaeizadeh and\n                  Piotr Milos and\n                  Blazej Osinski and\n                  Roy H. Campbell and\n                  Konrad Czechowski and\n                  Dumitru Erhan and\n                  Chelsea Finn and\n                  Piotr Kozakowski and\n                  Sergey Levine and\n                  Afroz Mohiuddin and\n                  Ryan Sepassi and\n                  George Tucker and\n                  Henryk Michalewski},\n  title        = {Model Based Reinforcement Learning for Atari},\n  booktitle    = {ICLR},\n  year         = {2020},\n\n}",
    "majorCategory": "奠基性工作",
    "subcategories": [
      "经典WM与模型式RL"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-ad5e94306036d2729d9f",
    "title": "Room-Across-Room: Multilingual Vision-and-Language Navigation with Dense Spatiotemporal Grounding",
    "authors": "Alexander Ku; Peter Anderson; Roma Patel; Eugene Ie; Jason Baldridge",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "rxr",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "EMNLP 2020",
    "paperUrl": "https://aclanthology.org/2020.emnlp-main.356/",
    "pdfUrl": "https://aclanthology.org/2020.emnlp-main.356.pdf",
    "doi": "https://doi.org/10.18653/v1/2020.emnlp-main.356",
    "publicationYear": 2020,
    "bibtex": "@inproceedings{rxr,\n  title={{Room-Across-Room}: Multilingual Vision-and-Language Navigation with Dense Spatiotemporal Grounding},\n  author={Alexander Ku and Peter Anderson and Roma Patel and Eugene Ie and Jason Baldridge},\n  booktitle={EMNLP},\n  year={2020}\n}",
    "majorCategory": "数据集",
    "subcategories": [
      "视觉语言导航数据",
      "多语言与时空标注",
      "基准与评测协议"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "综述明确"
  },
  {
    "id": "ref-a5d6a00ebc235c8fdf08",
    "title": "nuScenes: A Multimodal Dataset for Autonomous Driving",
    "authors": "Holger Caesar; Varun Bankiti; Alex H. Lang; Sourabh Vora; Venice Erin Liong; Qiang Xu; Anush Krishnan; Yu Pan; Giancarlo Baldan; Oscar Beijbom",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "nuscenes",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": "https://www.nuscenes.org/",
    "venue": "CVPR 2020",
    "paperUrl": "https://openaccess.thecvf.com/content_CVPR_2020/html/Caesar_nuScenes_A_Multimodal_Dataset_for_Autonomous_Driving_CVPR_2020_paper.html",
    "pdfUrl": "https://openaccess.thecvf.com/content_CVPR_2020/papers/Caesar_nuScenes_A_Multimodal_Dataset_for_Autonomous_Driving_CVPR_2020_paper.pdf",
    "doi": null,
    "publicationYear": 2020,
    "bibtex": "@inproceedings{nuscenes,\n  author       = {Holger Caesar and\n                  Varun Bankiti and\n                  Alex H. Lang and\n                  Sourabh Vora and\n                  Venice Erin Liong and\n                  Qiang Xu and\n                  Anush Krishnan and\n                  Yu Pan and\n                  Giancarlo Baldan and\n                  Oscar Beijbom},\n  title        = {nuScenes: {A} Multimodal Dataset for Autonomous Driving},\n  booktitle    = {CVPR},\n  pages        = {11618--11628},\n  year         = {2020},\n}",
    "majorCategory": "数据集",
    "subcategories": [
      "自动驾驶感知数据",
      "多传感器与空间标注",
      "基准与评测协议"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-a10caa76591cb7c581db",
    "title": "RLBench: The Robot Learning Benchmark & Learning Environment",
    "authors": "Stephen James; Zicong Ma; David Rovick Arrojo; Andrew J. Davison",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "rlbench",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "IEEE Robotics and Automation Letters",
    "paperUrl": "https://ieeexplore.ieee.org/document/9001253/",
    "pdfUrl": "https://arxiv.org/pdf/1909.12271",
    "doi": "https://doi.org/10.1109/lra.2020.2974707",
    "publicationYear": 2020,
    "bibtex": "@article{rlbench,\n  author       = {Stephen James and\n                  Zicong Ma and\n                  David Rovick Arrojo and\n                  Andrew J. Davison},\n  title        = {RLBench: The Robot Learning Benchmark {\\&} Learning Environment},\n  journal      = {RA-L},\n  volume       = {5},\n  number       = {2},\n  pages        = {3019--3026},\n  year         = {2020},\n}",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "机器人操作基准",
      "物理仿真",
      "示范数据"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-7147250a035b50dba3eb",
    "title": "Scalability in Perception for Autonomous Driving: Waymo Open Dataset",
    "authors": "Pei Sun; Henrik Kretzschmar; Xerxes Dotiwalla; Aurelien Chouard; Vijaysai Patnaik; Paul Tsui; James Guo; Yin Zhou; Yuning Chai; Benjamin Caine; Vijay Vasudevan; Wei Han; Jiquan Ngiam; Hang Zhao; Aleksei Timofeev; Scott Ettinger; Maxim Krivokon; Amy Gao; Aditya Joshi; Yu Zhang; Jonathon Shlens; Zhifeng Chen; Dragomir Anguelov",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "waymo",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "CVPR 2020",
    "paperUrl": "https://openaccess.thecvf.com/content_CVPR_2020/html/Sun_Scalability_in_Perception_for_Autonomous_Driving_Waymo_Open_Dataset_CVPR_2020_paper.html",
    "pdfUrl": "https://openaccess.thecvf.com/content_CVPR_2020/papers/Sun_Scalability_in_Perception_for_Autonomous_Driving_Waymo_Open_Dataset_CVPR_2020_paper.pdf",
    "doi": null,
    "publicationYear": 2020,
    "bibtex": "@InProceedings{waymo, author = {Sun, Pei and Kretzschmar, Henrik and Dotiwalla, Xerxes and Chouard, Aurelien and Patnaik, Vijaysai and Tsui, Paul and Guo, James and Zhou, Yin and Chai, Yuning and Caine, Benjamin and Vasudevan, Vijay and Han, Wei and Ngiam, Jiquan and Zhao, Hang and Timofeev, Aleksei and Ettinger, Scott and Krivokon, Maxim and Gao, Amy and Joshi, Aditya and Zhang, Yu and Shlens, Jonathon and Chen, Zhifeng and Anguelov, Dragomir}, title = {Scalability in Perception for Autonomous Driving: Waymo Open Dataset}, booktitle = {CVPR}, year = {2020} }",
    "majorCategory": "数据集",
    "subcategories": [
      "自动驾驶感知数据",
      "多传感器与空间标注",
      "基准与评测协议"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-663c359ac9bfe090d027",
    "title": "BDD100K: A Diverse Driving Dataset for Heterogeneous Multitask Learning",
    "authors": "Fisher Yu; Haofeng Chen; Xin Wang; Wenqi Xian; Yingying Chen; Fangchen Liu; Vashisht Madhavan; Trevor Darrell",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "bdd100k",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "CVPR 2020",
    "paperUrl": "https://openaccess.thecvf.com/content_CVPR_2020/html/Yu_BDD100K_A_Diverse_Driving_Dataset_for_Heterogeneous_Multitask_Learning_CVPR_2020_paper.html",
    "pdfUrl": "https://openaccess.thecvf.com/content_CVPR_2020/papers/Yu_BDD100K_A_Diverse_Driving_Dataset_for_Heterogeneous_Multitask_Learning_CVPR_2020_paper.pdf",
    "doi": null,
    "publicationYear": 2020,
    "bibtex": "@inproceedings{bdd100k,\n  author       = {Fisher Yu and\n                  Haofeng Chen and\n                  Xin Wang and\n                  Wenqi Xian and\n                  Yingying Chen and\n                  Fangchen Liu and\n                  Vashisht Madhavan and\n                  Trevor Darrell},\n  title        = {{BDD100K:} {A} Diverse Driving Dataset for Heterogeneous Multitask\n                  Learning},\n  booktitle    = {CVPR},\n  pages        = {2633--2642},\n  year         = {2020},\n}",
    "majorCategory": "数据集",
    "subcategories": [
      "自动驾驶感知数据",
      "多任务视觉标注",
      "基准与评测协议"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-626d1fe7536ab14c8fdc",
    "title": "Exploring the Limits of Transfer Learning with a Unified Text-to-Text Transformer",
    "authors": "Colin Raffel; Noam Shazeer; Adam Roberts; Katherine Lee; Sharan Narang; Michael Matena; Yanqi Zhou; Wei Li; Peter J. Liu",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "t5",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "Journal of Machine Learning Research",
    "paperUrl": "https://www.jmlr.org/beta/papers/v21/20-074.html",
    "pdfUrl": "https://jmlr.org/papers/volume21/20-074/20-074.pdf",
    "doi": null,
    "publicationYear": 2020,
    "bibtex": "@article{t5,\n  author       = {Colin Raffel and\n                  Noam Shazeer and\n                  Adam Roberts and\n                  Katherine Lee and\n                  Sharan Narang and\n                  Michael Matena and\n                  Yanqi Zhou and\n                  Wei Li and\n                  Peter J. Liu},\n  title        = {Exploring the Limits of Transfer Learning with a Unified Text-to-Text\n                  Transformer},\n  journal      = {JMLR},\n  volume       = {21},\n  pages        = {140:1--140:67},\n  year         = {2020},\n  \n}",
    "majorCategory": "WAM Components",
    "subcategories": [
      "Language & vision-language backbones",
      "语言与VLM Backbone"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-5c37def65669693c1520",
    "title": "Leveraging Procedural Generation to Benchmark Reinforcement Learning",
    "authors": "Karl Cobbe; Christopher Hesse; Jacob Hilton; John Schulman",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "procgen",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "ICML 2020",
    "paperUrl": "https://proceedings.mlr.press/v119/cobbe20a.html",
    "pdfUrl": "https://proceedings.mlr.press/v119/cobbe20a/cobbe20a.pdf",
    "doi": null,
    "publicationYear": 2020,
    "bibtex": "@inproceedings{procgen,\n  author       = {Karl Cobbe and\n                  Christopher Hesse and\n                  Jacob Hilton and\n                  John Schulman},\n  title        = {Leveraging Procedural Generation to Benchmark Reinforcement Learning},\n  booktitle    = {ICML},\n  pages        = {2048--2056},\n  year         = {2020},\n}",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "强化学习基准",
      "泛化评测",
      "物理仿真"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-4c7d069bbfa0f1875587",
    "title": "Meta-World: A Benchmark and Evaluation for Multi-Task and Meta Reinforcement Learning",
    "authors": "Tianhe Yu; Deirdre Quillen; Zhanpeng He; Ryan Julian; Karol Hausman; Chelsea Finn; Sergey Levine",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "metaworld",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "CoRL 2019",
    "paperUrl": "https://proceedings.mlr.press/v100/yu20a.html",
    "pdfUrl": "https://proceedings.mlr.press/v100/yu20a/yu20a.pdf",
    "doi": null,
    "publicationYear": 2020,
    "bibtex": "@inproceedings{metaworld,\n  author       = {Tianhe Yu and\n                  Deirdre Quillen and\n                  Zhanpeng He and\n                  Ryan Julian and\n                  Karol Hausman and\n                  Chelsea Finn and\n                  Sergey Levine},\n  title        = {Meta-World: {A} Benchmark and Evaluation for Multi-Task and Meta Reinforcement\n                  Learning},\n  booktitle    = {CoRL},\n  pages        = {1094--1100},\n  year         = {2019},\n}",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "机器人操作基准",
      "强化学习基准",
      "物理仿真"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-41759d2ff12db0233ba2",
    "title": "Language Models are Few-Shot Learners",
    "authors": "Tom Brown; Benjamin Mann; Nick Ryder; Melanie Subbiah; Jared D Kaplan; Prafulla Dhariwal; Arvind Neelakantan; Pranav Shyam; Girish Sastry; Amanda Askell; Sandhini Agarwal; Ariel Herbert-Voss; Gretchen Krueger; Tom Henighan; Rewon Child; Aditya Ramesh; Daniel Ziegler; Jeffrey Wu; Clemens Winter; Chris Hesse; Mark Chen; Eric Sigler; Mateusz Litwin; Scott Gray; Benjamin Chess; Jack Clark; Christopher Berner; Sam McCandlish; Alec Radford; Ilya Sutskever; Dario Amodei",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "gpt",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "NeurIPS 2020",
    "paperUrl": "https://proceedings.neurips.cc/paper_files/paper/2020/hash/1457c0d6bfcb4967418bfb8ac142f64a-Abstract.html",
    "pdfUrl": "https://arxiv.org/pdf/2005.14165",
    "doi": null,
    "publicationYear": 2020,
    "bibtex": "@inproceedings{gpt,\n author = {Brown, Tom and Mann, Benjamin and Ryder, Nick and Subbiah, Melanie and Kaplan, Jared D and Dhariwal, Prafulla and Neelakantan, Arvind and Shyam, Pranav and Sastry, Girish and Askell, Amanda and Agarwal, Sandhini and Herbert-Voss, Ariel and Krueger, Gretchen and Henighan, Tom and Child, Rewon and Ramesh, Aditya and Ziegler, Daniel and Wu, Jeffrey and Winter, Clemens and Hesse, Chris and Chen, Mark and Sigler, Eric and Litwin, Mateusz and Gray, Scott and Chess, Benjamin and Clark, Jack and Berner, Christopher and McCandlish, Sam and Radford, Alec and Sutskever, Ilya and Amodei, Dario},\n booktitle = {NeurIPS},\n pages = {1877--1901},\n title = {Language Models are Few-Shot Learners},\n volume = {33},\n year = {2020}\n}",
    "majorCategory": "WAM Components",
    "subcategories": [
      "Language & vision-language backbones",
      "语言与VLM Backbone"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-23813a5cbed0b2b6d37e",
    "title": "Dream to Control: Learning Behaviors by Latent Imagination",
    "authors": "Danijar Hafner; Timothy Lillicrap; Jimmy Ba; Mohammad Norouzi",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "dreamer",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "ICLR 2020",
    "paperUrl": "https://openreview.net/forum?id=S1lOTC4tDS",
    "pdfUrl": "https://openreview.net/pdf?id=S1lOTC4tDS",
    "doi": null,
    "publicationYear": 2020,
    "bibtex": "@inproceedings{dreamer,\n  author       = {Danijar Hafner and\n                  Timothy P. Lillicrap and\n                  Jimmy Ba and\n                  Mohammad Norouzi},\n  title        = {Dream to Control: Learning Behaviors by Latent Imagination},\n  booktitle    = {ICLR},\n  year         = {2020},\n \n}",
    "majorCategory": "奠基性工作",
    "subcategories": [
      "经典WM与模型式RL"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-21e4085744ceebdcf3ce",
    "title": "Behaviour Suite for Reinforcement Learning",
    "authors": "Ian Osband; Yotam Doron; Matteo Hessel; John Aslanides; Eren Sezener; Andre Saraiva; Katrina McKinney; Tor Lattimore; Csaba Szepesvári; Satinder Singh; Benjamin Van Roy; Richard S. Sutton; David Silver; Hado van Hasselt",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "bsuite",
    "arxivUrl": null,
    "codeUrls": [
      "https://github.com/deepmind/bsuite"
    ],
    "projectUrl": null,
    "venue": "ICLR 2020",
    "paperUrl": "https://iclr.cc/virtual/2020/poster/1492",
    "pdfUrl": "https://arxiv.org/pdf/1908.03568",
    "doi": null,
    "publicationYear": 2020,
    "bibtex": "@inproceedings{bsuite,\n  author       = {Ian Osband and\n                  Yotam Doron and\n                  Matteo Hessel and\n                  John Aslanides and\n                  Eren Sezener and\n                  Andre Saraiva and\n                  Katrina McKinney and\n                  Tor Lattimore and\n                  Csaba Szepesv{\\'{a}}ri and\n                  Satinder Singh and\n                  Benjamin Van Roy and\n                  Richard S. Sutton and\n                  David Silver and\n                  Hado van Hasselt},\n  title        = {Behaviour Suite for Reinforcement Learning},\n  booktitle    = {ICLR},\n  year         = {2020},\n}",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "强化学习基准",
      "评测协议与诊断"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-066817e565cc911bd5c1",
    "title": "Learning to summarize with human feedback",
    "authors": "Nisan Stiennon; Long Ouyang; Jeffrey Wu; Daniel Ziegler; Ryan Lowe; Chelsea Voss; Alec Radford; Dario Amodei; Paul F. Christiano",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "stiennon2020learning",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "NeurIPS 2020",
    "paperUrl": "https://papers.neurips.cc/paper_files/paper/2020/hash/1f89885d556929e98d3ef9b86448f951-Abstract.html",
    "pdfUrl": "https://proceedings.neurips.cc/paper_files/paper/2020/file/1f89885d556929e98d3ef9b86448f951-Paper.pdf",
    "doi": null,
    "publicationYear": 2020,
    "bibtex": "@article{stiennon2020learning,\n  title={Learning to summarize with human feedback},\n  author={Stiennon, Nisan and Ouyang, Long and Wu, Jeffrey and Ziegler, Daniel and Lowe, Ryan and Voss, Chelsea and Radford, Alec and Amodei, Dario and Christiano, Paul F},\n  journal={NeurIPS},\n  volume={33},\n  pages={3008--3021},\n  year={2020}\n}",
    "majorCategory": "奠基性工作",
    "subcategories": [
      "训练优化与蒸馏"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "综述明确"
  },
  {
    "id": "1910.11215",
    "title": "RoboNet: Large-Scale Multi-Robot Learning",
    "authors": "Sudeep Dasari; Frederik Ebert; Stephen Tian; Suraj Nair; Bernadette Bucher; Karl Schmeckpeper; Siddharth Singh; Sergey Levine; Chelsea Finn",
    "affiliations": "UC Berkeley; Stanford University; University of Pennsylvania; CMU",
    "contribution": "RoboNet pools robot experience to make visual control transferable. Pretraining improves adaptation with a few hundred target-robot trajectories, but relevant subsets can outperform the broader pool. Its central contribution is a shared dataset evaluated through two distinct control algorithms.",
    "abstract": "",
    "submittedDate": "2019-10-24",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "arxiv191011215",
    "arxivUrl": "https://arxiv.org/abs/1910.11215",
    "codeUrls": [],
    "projectUrl": null,
    "venue": "CoRL 2019",
    "paperUrl": "https://arxiv.org/abs/1910.11215",
    "pdfUrl": "https://arxiv.org/pdf/1910.11215",
    "doi": null,
    "publicationYear": 2019,
    "bibtex": "",
    "majorCategory": "数据集",
    "subcategories": [
      "跨机器人与多任务数据",
      "机器人交互数据"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": null
  },
  {
    "id": "1910.04867",
    "title": "A Large-scale Study of Representation Learning with the Visual Task Adaptation Benchmark",
    "authors": "Xiaohua Zhai; Joan Puigcerver; Alexander Kolesnikov; Pierre Ruyssen; Carlos Riquelme; Mario Lucic; Josip Djolonga; Andre Susano Pinto; Maxim Neumann; Alexey Dosovitskiy; Lucas Beyer; Olivier Bachem; Michael Tschannen; Marcin Michalski; Olivier Bousquet; Sylvain Gelly; Neil Houlsby",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2019-10-01",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "dmlab",
    "arxivUrl": "https://arxiv.org/abs/1910.04867",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/1910.04867",
    "pdfUrl": "https://arxiv.org/pdf/1910.04867",
    "doi": null,
    "publicationYear": 2019,
    "bibtex": "@article{dmlab,\n      title={The Visual Task Adaptation Benchmark}, \n      author={Xiaohua Zhai and Joan Puigcerver and Alexander Kolesnikov and Pierre Ruyssen and Carlos Riquelme and Mario Lucic and Josip Djolonga and Andre Susano Pinto and Maxim Neumann and Alexey Dosovitskiy and Lucas Beyer and Olivier Bachem and Michael Tschannen and Marcin Michalski and Olivier Bousquet and Sylvain Gelly and Neil Houlsby},\n      year={2019},\n      journal={arXiv:1910.04867},\n}",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "视觉表征迁移基准",
      "评测协议与诊断"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "1907.05446",
    "title": "General Evaluation for Instruction Conditioned Navigation using Dynamic Time Warping",
    "authors": "Gabriel Ilharco; Vihan Jain; Alexander Ku; Eugene Ie; Jason Baldridge",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2019-07-11",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "ndtw",
    "arxivUrl": "https://arxiv.org/abs/1907.05446",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/1907.05446",
    "pdfUrl": "https://arxiv.org/pdf/1907.05446",
    "doi": null,
    "publicationYear": 2019,
    "bibtex": "@article{ndtw,\n  title={General Evaluation for Instruction Conditioned Navigation using Dynamic Time Warping},\n  author={Ilharco, Gabriel and Jain, Vihan and Ku, Alexander and Ie, Eugene and Baldridge, Jason},\n  journal={arXiv:1907.05446},\n  year={2019}\n}",
    "majorCategory": "评估指标（Metrics）",
    "subcategories": [
      "导航评估指标",
      "评估指标与协议"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-e6aba3e8a3ffe044339d",
    "title": "When to Trust Your Model: Model-Based Policy Optimization",
    "authors": "Michael Janner; Justin Fu; Marvin Zhang; Sergey Levine",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "trust",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "NeurIPS 2019",
    "paperUrl": "https://proceedings.neurips.cc/paper_files/paper/2019/hash/5faf461eff3099671ad63c6f3f094f7f-Abstract.html",
    "pdfUrl": "https://proceedings.neurips.cc/paper/2019/file/5faf461eff3099671ad63c6f3f094f7f-Paper.pdf",
    "doi": null,
    "publicationYear": 2019,
    "bibtex": "@article{trust,\n  title={When to trust your model: Model-based policy optimization},\n  author={Janner, Michael and Fu, Justin and Zhang, Marvin and Levine, Sergey},\n  journal={NeurIPS},\n  volume={32},\n  year={2019}\n}",
    "majorCategory": "奠基性工作",
    "subcategories": [
      "经典WM与模型式RL",
      "训练优化与蒸馏"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "综述明确"
  },
  {
    "id": "ref-c7de129d1b2c440e37ac",
    "title": "LIC-Fusion: LiDAR-Inertial-Camera Odometry",
    "authors": "Xingxing Zuo; Patrick Geneva; Woosik Lee; Yong Liu; Guoquan Huang",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "Zuo2019LICFusionLiDARInertialCamera",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "IROS 2019",
    "paperUrl": "https://ieeexplore.ieee.org/document/8967746/",
    "pdfUrl": "https://arxiv.org/pdf/1909.04102",
    "doi": "https://doi.org/10.1109/iros40897.2019.8967746",
    "publicationYear": 2019,
    "bibtex": "@inproceedings{Zuo2019LICFusionLiDARInertialCamera,\n  author       = {Xingxing Zuo and\n                  Patrick Geneva and\n                  Woosik Lee and\n                  Yong Liu and\n                  Guoquan Huang},\n  title        = {LIC-Fusion: LiDAR-Inertial-Camera Odometry},\n  booktitle    = {IROS},\n  pages        = {5848--5854},\n  year         = {2019},\n\n}",
    "majorCategory": "Related resources",
    "subcategories": [
      "三维表示与状态估计"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-b06d4e2caeb9c57607b9",
    "title": "Habitat: A Platform for Embodied AI Research",
    "authors": "Manolis Savva; Jitendra Malik; Devi Parikh; Dhruv Batra; Abhishek Kadian; Oleksandr Maksymets; Yili Zhao; Erik Wijmans; Bhavana Jain; Julian Straub; Jia Liu; Vladlen Koltun",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "habitat",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "ICCV 2019",
    "paperUrl": "https://openaccess.thecvf.com/content_ICCV_2019/papers/Savva_Habitat_A_Platform_for_Embodied_AI_Research_ICCV_2019_paper.pdf",
    "pdfUrl": "https://openaccess.thecvf.com/content_ICCV_2019/papers/Savva_Habitat_A_Platform_for_Embodied_AI_Research_ICCV_2019_paper.pdf",
    "doi": null,
    "publicationYear": 2019,
    "bibtex": "@inproceedings{habitat,\n  author       = {Manolis Savva and\n                  Jitendra Malik and\n                  Devi Parikh and\n                  Dhruv Batra and\n                  Abhishek Kadian and\n                  Oleksandr Maksymets and\n                  Yili Zhao and\n                  Erik Wijmans and\n                  Bhavana Jain and\n                  Julian Straub and\n                  Jia Liu and\n                  Vladlen Koltun},\n  title        = {Habitat: {A} Platform for Embodied {AI} Research},\n  booktitle    = {ICCV},\n  pages        = {9338--9346},\n  year         = {2019},\n  \n}",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "物理仿真",
      "导航基准"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-901dd4e018709bd1ca32",
    "title": "Argoverse: 3D Tracking and Forecasting With Rich Maps",
    "authors": "Ming-Fang Chang; John Lambert; Patsorn Sangkloy; Jagjeet Singh; Slawomir Bak; Andrew Hartnett; De Wang; Peter Carr; Simon Lucey; Deva Ramanan; James Hays",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "argoverse",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "CVPR 2019",
    "paperUrl": "https://openaccess.thecvf.com/content_CVPR_2019/html/Chang_Argoverse_3D_Tracking_and_Forecasting_With_Rich_Maps_CVPR_2019_paper.html",
    "pdfUrl": "https://openaccess.thecvf.com/content_CVPR_2019/papers/Chang_Argoverse_3D_Tracking_and_Forecasting_With_Rich_Maps_CVPR_2019_paper.pdf",
    "doi": null,
    "publicationYear": 2019,
    "bibtex": "@inproceedings{argoverse,\n  author       = {Ming{-}Fang Chang and\n                  John Lambert and\n                  Patsorn Sangkloy and\n                  Jagjeet Singh and\n                  Slawomir Bak and\n                  Andrew Hartnett and\n                  De Wang and\n                  Peter Carr and\n                  Simon Lucey and\n                  Deva Ramanan and\n                  James Hays},\n  title        = {Argoverse: 3D Tracking and Forecasting With Rich Maps},\n  booktitle    = {CVPR},\n  pages        = {8748--8757},\n  year         = {2019},\n}",
    "majorCategory": "数据集",
    "subcategories": [
      "自动驾驶感知数据",
      "运动预测与规划数据",
      "基准与评测协议"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-8c0a34d0c6ab6b64e3ae",
    "title": "Diversity is all you need: Learning skills without a reward function",
    "authors": "Benjamin Eysenbach; Abhishek Gupta; Julian Ibarz; Sergey Levine",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "Benjamin_2019",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "ICLR 2019",
    "paperUrl": "https://iclr.cc/virtual/2019/poster/720",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": 2019,
    "bibtex": "@conference{Benjamin_2019,\ntitle = \"Diversity is all you need: Learning skills without a reward function\",\nauthor = \"Benjamin Eysenbach and Julian Ibarz and Abhishek Gupta and Sergey Levine\",\nbooktitle={ICLR},\nyear = \"2019\",\n\n}",
    "majorCategory": "奠基性工作",
    "subcategories": [
      "动作策略基础",
      "理论与规划"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "综述明确"
  },
  {
    "id": "ref-887c80067c952c4655d9",
    "title": "HowTo100M: Learning a Text-Video Embedding by Watching Hundred Million Narrated Video Clips",
    "authors": "Antoine Miech; Dimitri Zhukov; Jean-Baptiste Alayrac; Makarand Tapaswi; Ivan Laptev; Josef Sivic",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "howto100m",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "ICCV 2019",
    "paperUrl": "https://openaccess.thecvf.com/content_ICCV_2019/papers/Miech_HowTo100M_Learning_a_Text-Video_Embedding_by_Watching_Hundred_Million_Narrated_ICCV_2019_paper.pdf",
    "pdfUrl": "https://openaccess.thecvf.com/content_ICCV_2019/papers/Miech_HowTo100M_Learning_a_Text-Video_Embedding_by_Watching_Hundred_Million_Narrated_ICCV_2019_paper.pdf",
    "doi": null,
    "publicationYear": 2019,
    "bibtex": "@inproceedings{howto100m,\n  author       = {Antoine Miech and\n                  Dimitri Zhukov and\n                  Jean{-}Baptiste Alayrac and\n                  Makarand Tapaswi and\n                  Ivan Laptev and\n                  Josef Sivic},\n  title        = {HowTo100M: Learning a Text-Video Embedding by Watching Hundred Million\n                  Narrated Video Clips},\n  booktitle    = {ICCV},\n  pages        = {2630--2640},\n  year         = {2019},\n\n}",
    "majorCategory": "数据集",
    "subcategories": [
      "视频语言预训练数据",
      "语言标注与再标注",
      "视频动作理解数据"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-74afd84d073d16a0a3a1",
    "title": "Learning Latent Dynamics for Planning from Pixels",
    "authors": "Danijar Hafner; Timothy Lillicrap; Ian Fischer; Ruben Villegas; David Ha; Honglak Lee; James Davidson",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "hafner2019learning",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "ICML 2019",
    "paperUrl": "https://proceedings.mlr.press/v97/hafner19a.html",
    "pdfUrl": "https://arxiv.org/pdf/1811.04551",
    "doi": null,
    "publicationYear": 2019,
    "bibtex": "@inproceedings{hafner2019learning,\n  author       = {Danijar Hafner and\n                  Timothy P. Lillicrap and\n                  Ian Fischer and\n                  Ruben Villegas and\n                  David Ha and\n                  Honglak Lee and\n                  James Davidson},\n  title        = {Learning Latent Dynamics for Planning from Pixels},\n  booktitle    = {ICML},\n  pages        = {2555--2565},\n  year         = {2019},\n}",
    "majorCategory": "奠基性工作",
    "subcategories": [
      "经典WM与模型式RL"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "1812.01717",
    "title": "Towards Accurate Generative Models of Video: A New Metric & Challenges",
    "authors": "Thomas Unterthiner; Sjoerd van Steenkiste; Karol Kurach; Raphael Marinier; Marcin Michalski; Sylvain Gelly",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2018-12-03",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "fvd",
    "arxivUrl": "https://arxiv.org/abs/1812.01717",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/1812.01717",
    "pdfUrl": "https://arxiv.org/pdf/1812.01717",
    "doi": null,
    "publicationYear": 2018,
    "bibtex": "@article{fvd,\n  title={Towards Accurate Generative Models of Video: A New Metric {\\&} Challenges},\n  author={Unterthiner, Thomas and van Steenkiste, Sjoerd and Kurach, Karol and Marinier, Raphael and Michalski, Marcin and Gelly, Sylvain},\n  journal={arXiv:1812.01717},\n  year={2018}\n}",
    "majorCategory": "评估指标（Metrics）",
    "subcategories": [
      "视频质量指标",
      "评估指标与协议",
      "基准与模拟器"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "1807.06757",
    "title": "On Evaluation of Embodied Navigation Agents",
    "authors": "Peter Anderson; Angel Chang; Devendra Singh Chaplot; Alexey Dosovitskiy; Saurabh Gupta; Vladlen Koltun; Jana Kosecka; Jitendra Malik; Roozbeh Mottaghi; Manolis Savva; Amir R. Zamir",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2018-07-18",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "spl_eval",
    "arxivUrl": "https://arxiv.org/abs/1807.06757",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/1807.06757",
    "pdfUrl": "https://arxiv.org/pdf/1807.06757",
    "doi": null,
    "publicationYear": 2018,
    "bibtex": "@article{spl_eval,\n  title={On Evaluation of Embodied Navigation Agents},\n  author={Anderson, Peter and Chang, Angel X. and Chaplot, Devendra Singh and Dosovitskiy, Alexey and Gupta, Saurabh and Koltun, Vladlen and Kosecka, Jana and Malik, Jitendra and Mottaghi, Roozbeh and Savva, Manolis and Zamir, Amir R.},\n  journal={arXiv:1807.06757},\n  year={2018}\n}",
    "majorCategory": "评估指标（Metrics）",
    "subcategories": [
      "导航评估指标",
      "评估指标与协议"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "1803.10122",
    "title": "World Models",
    "authors": "David Ha; Jürgen Schmidhuber",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2018-03-27",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "world_model",
    "arxivUrl": "https://arxiv.org/abs/1803.10122",
    "codeUrls": [],
    "projectUrl": "https://worldmodels.github.io/",
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/1803.10122",
    "pdfUrl": "https://arxiv.org/pdf/1803.10122",
    "doi": "https://doi.org/10.5281/zenodo.1207631",
    "publicationYear": 2018,
    "bibtex": "@article{world_model,\n      title={World models}, \n      author={Ha, David and Schmidhuber, J{\\\"u}rgen},\n      year={2018},\n      journal={arXiv:1803.10122},\n}",
    "majorCategory": "奠基性工作",
    "subcategories": [
      "经典WM与模型式RL"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "1801.00690",
    "title": "DeepMind Control Suite",
    "authors": "Yuval Tassa; Yotam Doron; Alistair Muldal; Tom Erez; Yazhe Li; Diego de Las Casas; David Budden; Abbas Abdolmaleki; Josh Merel; Andrew Lefrancq; Timothy Lillicrap; Martin Riedmiller",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2018-01-02",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "dmcontrol",
    "arxivUrl": "https://arxiv.org/abs/1801.00690",
    "codeUrls": [
      "https://github.com/google-deepmind/dm_control"
    ],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/1801.00690",
    "pdfUrl": "https://arxiv.org/pdf/1801.00690",
    "doi": null,
    "publicationYear": 2018,
    "bibtex": "@article{dmcontrol,\n      title={DeepMind Control Suite}, \n      author={Yuval Tassa and Yotam Doron and Alistair Muldal and Tom Erez and Yazhe Li and Diego de Las Casas and David Budden and Abbas Abdolmaleki and Josh Merel and Andrew Lefrancq and Timothy Lillicrap and Martin Riedmiller},\n      year={2018},\n      journal={arXiv:1801.00690},\n}",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "强化学习基准",
      "物理仿真"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-e0201a8da35e8b32ce42",
    "title": "Vision-and-Language Navigation: Interpreting Visually-Grounded Navigation Instructions in Real Environments",
    "authors": "Peter Anderson; Qi Wu; Damien Teney; Jake Bruce; Mark Johnson; Niko Sünderhauf; Ian D. Reid; Stephen Gould; Anton van den Hengel",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "r2r",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "CVPR 2018",
    "paperUrl": "https://openaccess.thecvf.com/content_cvpr_2018/html/Anderson_Vision-and-Language_Navigation_Interpreting_CVPR_2018_paper.html",
    "pdfUrl": "https://openaccess.thecvf.com/content_cvpr_2018/papers/Anderson_Vision-and-Language_Navigation_Interpreting_CVPR_2018_paper.pdf",
    "doi": null,
    "publicationYear": 2018,
    "bibtex": "@inproceedings{r2r,\n  author       = {Peter Anderson and\n                  Qi Wu and\n                  Damien Teney and\n                  Jake Bruce and\n                  Mark Johnson and\n                  Niko S{\\\"{u}}nderhauf and\n                  Ian D. Reid and\n                  Stephen Gould and\n                  Anton van den Hengel},\n  title        = {Vision-and-Language Navigation: Interpreting Visually-Grounded Navigation\n                  Instructions in Real Environments},\n  booktitle    = {CVPR},\n  pages        = {3674--3683},\n  year         = {2018},\n}",
    "majorCategory": "数据集",
    "subcategories": [
      "视觉语言导航数据",
      "语言标注与再标注",
      "基准与评测协议"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-beec1282930b73cf1272",
    "title": "The Unreasonable Effectiveness of Deep Features as a Perceptual Metric",
    "authors": "Richard Zhang; Phillip Isola; Alexei A. Efros; Eli Shechtman; Oliver Wang",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "lpips",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "CVPR 2018",
    "paperUrl": "https://openaccess.thecvf.com/content_cvpr_2018/html/Zhang_The_Unreasonable_Effectiveness_CVPR_2018_paper.html",
    "pdfUrl": "https://openaccess.thecvf.com/content_cvpr_2018/papers/Zhang_The_Unreasonable_Effectiveness_CVPR_2018_paper.pdf",
    "doi": null,
    "publicationYear": 2018,
    "bibtex": "@inproceedings{lpips,\n  title={The Unreasonable Effectiveness of Deep Features as a Perceptual Metric},\n  author={Zhang, Richard and Isola, Phillip and Efros, Alexei A. and Shechtman, Eli and Wang, Oliver},\n  booktitle={CVPR},\n  pages={586--595},\n  year={2018}\n}",
    "majorCategory": "评估指标（Metrics）",
    "subcategories": [
      "图像质量指标",
      "评估指标与协议",
      "视觉编码器与表征"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-a3e337f84bff0522e5e1",
    "title": "Sim-to-Real Transfer of Robotic Control with Dynamics Randomization",
    "authors": "Xue Bin Peng; Marcin Andrychowicz; Wojciech Zaremba; Pieter Abbeel",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "peng2017simtoreal",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "ICRA 2018",
    "paperUrl": "https://ieeexplore.ieee.org/document/8460528",
    "pdfUrl": "https://arxiv.org/pdf/1710.06537",
    "doi": "https://doi.org/10.1109/ICRA.2018.8460528",
    "publicationYear": 2018,
    "bibtex": "@inproceedings{peng2017simtoreal,\n  author       = {Xue Bin Peng and\n                  Marcin Andrychowicz and\n                  Wojciech Zaremba and\n                  Pieter Abbeel},\n  title        = {Sim-to-Real Transfer of Robotic Control with Dynamics Randomization},\n  booktitle    = {ICRA},\n  pages        = {1--8},\n  year         = {2018},\n}",
    "majorCategory": "奠基性工作",
    "subcategories": [
      "训练优化与蒸馏",
      "动作策略基础"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "1705.06950",
    "title": "The Kinetics Human Action Video Dataset",
    "authors": "Will Kay; Joao Carreira; Karen Simonyan; Brian Zhang; Chloe Hillier; Sudheendra Vijayanarasimhan; Fabio Viola; Tim Green; Trevor Back; Paul Natsev; Mustafa Suleyman; Andrew Zisserman",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": "2017-05-19",
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "kinetics",
    "arxivUrl": "https://arxiv.org/abs/1705.06950",
    "codeUrls": [],
    "projectUrl": null,
    "venue": null,
    "paperUrl": "https://arxiv.org/abs/1705.06950",
    "pdfUrl": "https://arxiv.org/pdf/1705.06950",
    "doi": null,
    "publicationYear": 2017,
    "bibtex": "@article{kinetics,\n      title={The Kinetics Human Action Video Dataset}, \n      author={Will Kay and Joao Carreira and Karen Simonyan and Brian Zhang and Chloe Hillier and Sudheendra Vijayanarasimhan and Fabio Viola and Tim Green and Trevor Back and Paul Natsev and Mustafa Suleyman and Andrew Zisserman},\n      year={2017},\n      journal={arXiv:1705.06950},\n}",
    "majorCategory": "数据集",
    "subcategories": [
      "视频动作理解数据",
      "类别与动作标注"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-eff2480dedeaf25358c4",
    "title": "Matterport3D: Learning from RGB-D Data in Indoor Environments",
    "authors": "Angel X. Chang; Angela Dai; Thomas A. Funkhouser; Maciej Halber; Matthias Nießner; Manolis Savva; Shuran Song; Andy Zeng; Yinda Zhang",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "matterport3d",
    "arxivUrl": null,
    "codeUrls": [
      "https://github.com/niessner/Matterport"
    ],
    "projectUrl": null,
    "venue": "3DV 2017",
    "paperUrl": "https://ieeexplore.ieee.org/document/8374622/",
    "pdfUrl": "https://arxiv.org/pdf/1709.06158",
    "doi": "https://doi.org/10.1109/3DV.2017.00081",
    "publicationYear": 2017,
    "bibtex": "@inproceedings{matterport3d,\n  author       = {Angel X. Chang and\n                  Angela Dai and\n                  Thomas A. Funkhouser and\n                  Maciej Halber and\n                  Matthias Nie{\\ss}ner and\n                  Manolis Savva and\n                  Shuran Song and\n                  Andy Zeng and\n                  Yinda Zhang},\n  title        = {Matterport3D: Learning from {RGB-D} Data in Indoor Environments},\n  booktitle    = {3DV},\n  pages        = {667--676},\n  year         = {2017},\n}",
    "majorCategory": "数据集",
    "subcategories": [
      "三维场景与RGB-D数据",
      "多传感器与空间标注"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-dcea01617aaebb6971e4",
    "title": "Factor Graphs for Robot Perception",
    "authors": "Frank Dellaert; Michael Kaess",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "Dellaert2017FactorGraphs; Dellaert_2017",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "Foundations and Trends in Robotics",
    "paperUrl": "https://www.cs.cmu.edu/~kaess/pub/Dellaert17fnt.html",
    "pdfUrl": "https://publications.ri.cmu.edu/storage/publications/2018/05/Dellaert17fnt.pdf",
    "doi": "https://doi.org/10.1561/2300000043",
    "publicationYear": 2017,
    "bibtex": "@article{Dellaert2017FactorGraphs,\n  author       = {Frank Dellaert and\n                  Michael Kaess},\n  title        = {Factor Graphs for Robot Perception},\n  journal      = {Found. Trends Robot.},\n  volume       = {6},\n  number       = {1-2},\n  pages        = {1--139},\n  year         = {2017},\n\n}\n\n@article{Dellaert_2017,\n  title     = {Factor Graphs for Robot Perception},\n  volume    = {6},\n  issn      = {1935-8261},\n  journal   = {Foundations and Trends in Robotics},\n  publisher = {Emerald},\n  author    = {Dellaert, Frank and Kaess, Michael},\n  year      = {2017},\n  pages     = {1–139}\n}",
    "majorCategory": "奠基性工作",
    "subcategories": [
      "三维表示与状态估计",
      "理论与规划"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "综述明确"
  },
  {
    "id": "ref-dacd62642bb5d66abc0c",
    "title": "Neural Discrete Representation Learning",
    "authors": "Aaron van den Oord; Oriol Vinyals; Koray Kavukcuoglu",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "vqvae",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "NeurIPS 2017",
    "paperUrl": "https://proceedings.neurips.cc/paper/2017/hash/7a98af17e63a0ac09ce2e96d03992fbc-Abstract.html",
    "pdfUrl": "https://arxiv.org/pdf/1711.00937",
    "doi": null,
    "publicationYear": 2017,
    "bibtex": "@article{vqvae,\n  title={Neural discrete representation learning},\n  author={Van Den Oord, Aaron and Vinyals, Oriol and others},\n  journal={NeurIPS},\n  volume={30},\n  year={2017}\n}",
    "majorCategory": "WAM Components",
    "subcategories": [
      "Generative modeling & tokenizers",
      "视觉编码器与表征"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-c71dcf53130e8fdc61fa",
    "title": "The \"Something Something\" Video Database for Learning and Evaluating Visual Common Sense",
    "authors": "Raghav Goyal; Samira Ebrahimi Kahou; Vincent Michalski; Joanna Materzynska; Susanne Westphal; Heuna Kim; Valentin Haenel; Ingo Fruend; Peter Yianilos; Moritz Mueller-Freitag; Florian Hoppe; Christian Thurau; Ingo Bax; Roland Memisevic",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "ssv2",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "ICCV 2017",
    "paperUrl": "https://openaccess.thecvf.com/content_iccv_2017/html/Goyal_The_Something_Something_ICCV_2017_paper.html",
    "pdfUrl": "https://openaccess.thecvf.com/content_ICCV_2017/papers/Goyal_The_Something_Something_ICCV_2017_paper.pdf",
    "doi": null,
    "publicationYear": 2017,
    "bibtex": "@inproceedings{ssv2,\n  title={The\" something something\" video database for learning and evaluating visual common sense},\n  author={Goyal, Raghav and Ebrahimi Kahou, Samira and Michalski, Vincent and Materzynska, Joanna and Westphal, Susanne and Kim, Heuna and Haenel, Valentin and Fruend, Ingo and Yianilos, Peter and Mueller-Freitag, Moritz and others},\n  booktitle={ICCV},\n  pages={5842--5850},\n  year={2017}\n}",
    "majorCategory": "数据集",
    "subcategories": [
      "视频动作理解数据",
      "语言标注与再标注"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-b3c87355da574d40991f",
    "title": "GANs Trained by a Two Time-Scale Update Rule Converge to a Local Nash Equilibrium",
    "authors": "Martin Heusel; Hubert Ramsauer; Thomas Unterthiner; Bernhard Nessler; Sepp Hochreiter",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "fid",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "NeurIPS 2017",
    "paperUrl": "https://proceedings.neurips.cc/paper_files/paper/2017/hash/8a1d694707eb0fefe65871369074926d-Abstract.html",
    "pdfUrl": "https://proceedings.neurips.cc/paper_files/paper/2017/file/8a1d694707eb0fefe65871369074926d-Paper.pdf",
    "doi": null,
    "publicationYear": 2017,
    "bibtex": "@inproceedings{fid,\n  title={{GANs} Trained by a Two Time-Scale Update Rule Converge to a Local Nash Equilibrium},\n  author={Heusel, Martin and Ramsauer, Hubert and Unterthiner, Thomas and Nessler, Bernhard and Hochreiter, Sepp},\n  booktitle={NeurIPS},\n  year={2017}\n}",
    "majorCategory": "评估指标（Metrics）",
    "subcategories": [
      "图像质量指标",
      "评估指标与协议",
      "训练优化与蒸馏"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "综述明确"
  },
  {
    "id": "ref-a6e65f34d4c161a2ba59",
    "title": "PointNet: Deep Learning on Point Sets for 3D Classification and Segmentation",
    "authors": "Charles R. Qi; Hao Su; Kaichun Mo; Leonidas J. Guibas",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "qi2017pointnet",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "CVPR 2017",
    "paperUrl": "https://openaccess.thecvf.com/content_cvpr_2017/html/Qi_PointNet_Deep_Learning_CVPR_2017_paper.html",
    "pdfUrl": "https://openaccess.thecvf.com/content_cvpr_2017/papers/Qi_PointNet_Deep_Learning_CVPR_2017_paper.pdf",
    "doi": null,
    "publicationYear": 2017,
    "bibtex": "@inproceedings{qi2017pointnet,\n  author       = {Charles Ruizhongtai Qi and\n                  Hao Su and\n                  Kaichun Mo and\n                  Leonidas J. Guibas},\n  title        = {PointNet: Deep Learning on Point Sets for 3D Classification and Segmentation},\n  booktitle    = {CVPR},\n  pages        = {77--85},\n  year         = {2017},\n \n}",
    "majorCategory": "WAM Components",
    "subcategories": [
      "Spatial perception & geometry",
      "视觉编码器与表征",
      "三维表示与状态估计"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-75baf2ba00d451202231",
    "title": "CARLA: An Open Urban Driving Simulator",
    "authors": "Alexey Dosovitskiy; Germán Ros; Felipe Codevilla; Antonio M. López; Vladlen Koltun",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "carla",
    "arxivUrl": null,
    "codeUrls": [
      "https://github.com/carla-simulator/carla"
    ],
    "projectUrl": null,
    "venue": "CoRL 2017",
    "paperUrl": "https://proceedings.mlr.press/v78/dosovitskiy17a.html",
    "pdfUrl": "https://proceedings.mlr.press/v78/dosovitskiy17a/dosovitskiy17a.pdf",
    "doi": null,
    "publicationYear": 2017,
    "bibtex": "@inproceedings{carla,\n  author       = {Alexey Dosovitskiy and\n                  Germ{\\'{a}}n Ros and\n                  Felipe Codevilla and\n                  Antonio M. L{\\'{o}}pez and\n                  Vladlen Koltun},\n  title        = {{CARLA:} An Open Urban Driving Simulator},\n  booktitle    = {CoRL},\n  pages        = {1--16},\n  year         = {2017},\n}",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "物理仿真",
      "自动驾驶基准"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-2062ffd7f6397312bd01",
    "title": "Deep Reinforcement Learning from Human Preferences",
    "authors": "Paul F. Christiano; Jan Leike; Tom Brown; Miljan Martic; Shane Legg; Dario Amodei",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "christiano2017deep",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "NeurIPS 2017",
    "paperUrl": "https://proceedings.neurips.cc/paper/2017/hash/d5e2c0adad503c91f91df240d0cd4e49-Abstract.html",
    "pdfUrl": "https://proceedings.neurips.cc/paper/2017/file/d5e2c0adad503c91f91df240d0cd4e49-Paper.pdf",
    "doi": null,
    "publicationYear": 2017,
    "bibtex": "@article{christiano2017deep,\n  title={Deep reinforcement learning from human preferences},\n  author={Christiano, Paul F and Leike, Jan and Brown, Tom and Martic, Miljan and Legg, Shane and Amodei, Dario},\n  journal={NeurIPS},\n  volume={30},\n  year={2017}\n}",
    "majorCategory": "奠基性工作",
    "subcategories": [
      "训练优化与蒸馏"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "综述明确"
  },
  {
    "id": "ref-24ca74d7b1d59b2ae206",
    "title": "CIDEr: Consensus-Based Image Description Evaluation",
    "authors": "Ramakrishna Vedantam; C. Lawrence Zitnick; Devi Parikh",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "cider",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "CVPR 2015",
    "paperUrl": "https://openaccess.thecvf.com/content_cvpr_2015/html/Vedantam_CIDEr_Consensus-Based_Image_2015_CVPR_paper.html",
    "pdfUrl": "https://openaccess.thecvf.com/content_cvpr_2015/papers/Vedantam_CIDEr_Consensus-Based_Image_2015_CVPR_paper.pdf",
    "doi": null,
    "publicationYear": 2015,
    "bibtex": "@inproceedings{cider,\n  title={{CIDEr}: Consensus-Based Image Description Evaluation},\n  author={Vedantam, Ramakrishna and Zitnick, C. Lawrence and Parikh, Devi},\n  booktitle={CVPR},\n  pages={4566--4575},\n  year={2015}\n}",
    "majorCategory": "评估指标（Metrics）",
    "subcategories": [
      "语言生成指标",
      "评估指标与协议"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "综述明确"
  },
  {
    "id": "ref-b5a6c332a7beaf120245",
    "title": "Advances and applications of occupancy models",
    "authors": "Larissa L Bailey; Darryl I MacKenzie; James D Nichols",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "occupancy",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "Methods in Ecology and Evolution 5(12), 2014",
    "paperUrl": "https://besjournals.onlinelibrary.wiley.com/doi/10.1111/2041-210X.12100",
    "pdfUrl": null,
    "doi": "https://doi.org/10.1111/2041-210X.12100",
    "publicationYear": 2014,
    "bibtex": "@article{occupancy,\n  title={Advances and applications of occupancy models},\n  author={Bailey, Larissa L and MacKenzie, Darryl I and Nichols, James D},\n  journal={Methods Ecol. Evol.},\n  volume={5},\n  number={12},\n  pages={1269--1279},\n  year={2014},\n}",
    "majorCategory": null,
    "subcategories": [
      "综述与技术资源"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "1312.6114",
    "title": "Auto-Encoding Variational Bayes",
    "authors": "Diederik P Kingma; Max Welling",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "kingma2013auto",
    "arxivUrl": "https://arxiv.org/abs/1312.6114",
    "codeUrls": [],
    "projectUrl": null,
    "venue": "ICLR 2014",
    "paperUrl": "https://arxiv.org/abs/1312.6114",
    "pdfUrl": "https://arxiv.org/pdf/1312.6114",
    "doi": null,
    "publicationYear": 2014,
    "bibtex": "@inproceedings{kingma2013auto,\n      title={Auto-encoding variational bayes}, \n      author={Kingma, Diederik P and Welling, Max},\n      booktitle    = {ICLR},\n      year={2014},\n}",
    "majorCategory": "WAM Components",
    "subcategories": [
      "Generative modeling & tokenizers",
      "视觉编码器与表征"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-147718b7415fa449b40e",
    "title": "The Arcade Learning Environment: An Evaluation Platform for General Agents",
    "authors": "Marc G. Bellemare; Yavar Naddaf; Joel Veness; Michael Bowling",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "atari",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "Journal of Artificial Intelligence Research",
    "paperUrl": "https://jair.org/index.php/jair/article/view/10819",
    "pdfUrl": "https://arxiv.org/pdf/1207.4708",
    "doi": "https://doi.org/10.1613/jair.3912",
    "publicationYear": 2013,
    "bibtex": "@article{atari,\n  author       = {Marc G. Bellemare and\n                  Yavar Naddaf and\n                  Joel Veness and\n                  Michael Bowling},\n  title        = {The Arcade Learning Environment: An Evaluation Platform for General\n                  Agents},\n  journal      = {JAIR},\n  volume       = {47},\n  pages        = {253--279},\n  year         = {2013},\n}",
    "majorCategory": "评测基准与模拟器",
    "subcategories": [
      "强化学习基准",
      "游戏评测环境",
      "评测协议与诊断"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "一手资料核实"
  },
  {
    "id": "ref-d0e3b6e2a926a91ba6c0",
    "title": "A benchmark for the evaluation of RGB-D SLAM systems",
    "authors": "Jrgen Sturm; Nikolas Engelhard; Felix Endres; Wolfram Burgard; Daniel Cremers",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "tum_rgbd",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": "https://cvg.cit.tum.de/data/datasets/rgbd-dataset",
    "venue": "IROS 2012",
    "paperUrl": "https://portal.fis.tum.de/en/publications/a-benchmark-for-the-evaluation-of-rgb-d-slam-systems/",
    "pdfUrl": "https://cvai.cit.tum.de/_media/spezial/bib/sturm12iros.pdf",
    "doi": "https://doi.org/10.1109/IROS.2012.6385773",
    "publicationYear": 2012,
    "bibtex": "@inproceedings{tum_rgbd,\n  title={A Benchmark for the Evaluation of {RGB-D} {SLAM} Systems},\n  author={Sturm, J{\\\"u}rgen and Engelhard, Nikolas and Endres, Felix and Burgard, Wolfram and Cremers, Daniel},\n  booktitle={IROS},\n  pages={573--580},\n  year={2012}\n}",
    "majorCategory": "数据集",
    "subcategories": [
      "三维场景与RGB-D数据",
      "定位与SLAM数据",
      "基准与评测协议"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "综述明确"
  },
  {
    "id": "ref-46f2622e9aef7e722cb2",
    "title": "Are we ready for autonomous driving? The KITTI vision benchmark suite",
    "authors": "Andreas Geiger; Philip Lenz; Raquel Urtasun",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "kitti",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": "https://www.cvlibs.net/datasets/kitti",
    "venue": "CVPR 2012",
    "paperUrl": "https://ieeexplore.ieee.org/document/6248074",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": 2012,
    "bibtex": "@inproceedings{kitti,\n  author       = {Andreas Geiger and\n                  Philip Lenz and\n                  Raquel Urtasun},\n  title        = {Are we ready for autonomous driving? The {KITTI} vision benchmark\n                  suite},\n  booktitle    = {CVPR},\n  pages        = {3354--3361},\n  year         = {2012},\n}",
    "majorCategory": "数据集",
    "subcategories": [
      "自动驾驶感知数据",
      "定位与SLAM数据",
      "基准与评测协议"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "综述明确"
  },
  {
    "id": "ref-f81b18b1d9818e0e2585",
    "title": "Monte-Carlo Planning in Large POMDPs",
    "authors": "David Silver; Joel Veness",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "silver2010monte",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "NeurIPS 2010",
    "paperUrl": "https://proceedings.neurips.cc/paper/2010/hash/edfbe1afcf9246bb0d40eb4d8027d90f-Abstract.html",
    "pdfUrl": "https://proceedings.neurips.cc/paper/2010/file/edfbe1afcf9246bb0d40eb4d8027d90f-Paper.pdf",
    "doi": null,
    "publicationYear": 2010,
    "bibtex": "@inproceedings{silver2010monte,\n  author       = {David Silver and\n                  Joel Veness},\n  title        = {Monte-Carlo Planning in Large POMDPs},\n  booktitle    = {NeurIPS},\n  pages        = {2164--2172},\n  year         = {2010},\n\n}",
    "majorCategory": "奠基性工作",
    "subcategories": [
      "理论与规划"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "综述明确"
  },
  {
    "id": "ref-73844778b2bcfa68b4e9",
    "title": "METEOR: An Automatic Metric for MT Evaluation with Improved Correlation with Human Judgments",
    "authors": "Satanjeev Banerjee; Alon Lavie",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "meteor",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "ACL Workshop on Intrinsic and Extrinsic Evaluation Measures for Machine Translation and/or Summarization, 2005",
    "paperUrl": "https://aclanthology.org/W05-0909/",
    "pdfUrl": "https://aclanthology.org/W05-0909.pdf",
    "doi": null,
    "publicationYear": 2005,
    "bibtex": "@inproceedings{meteor,\n  title={{METEOR}: An Automatic Metric for MT Evaluation with Improved Correlation with Human Judgments},\n  author={Banerjee, Satanjeev and Lavie, Alon},\n  booktitle={ACL Workshop on Intrinsic and Extrinsic Evaluation Measures for Machine Translation and/or Summarization},\n  pages={65--72},\n  year={2005}\n}",
    "majorCategory": "评估指标（Metrics）",
    "subcategories": [
      "语言生成指标",
      "评估指标与协议"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "综述明确"
  },
  {
    "id": "ref-47f59ffa32a9f466d486",
    "title": "Probabilistic Robotics",
    "authors": "Sebastian Thrun; Wolfram Burgard; Dieter Fox",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "thrun2005probabilistic",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "MIT Press（图书）",
    "paperUrl": "https://mitpress.mit.edu/9780262201629/probabilistic-robotics/",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": 2005,
    "bibtex": "@book{thrun2005probabilistic,\n  author    = {Thrun, Sebastian and Burgard, Wolfram and Fox, Dieter},\n  title     = {Probabilistic Robotics},\n  year      = {2005},\n  publisher = {MIT Press},\n  address   = {Cambridge, Mass.},\n  isbn      = {978-0262201629},\n  location  = {Cambridge, MA, USA}\n}",
    "majorCategory": "奠基性工作",
    "subcategories": [
      "理论与规划",
      "三维表示与状态估计"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "综述明确"
  },
  {
    "id": "ref-0276752be34087981228",
    "title": "Image Quality Assessment: From Error Visibility to Structural Similarity",
    "authors": "Zhou Wang; Alan C. Bovik; Hamid R. Sheikh; Eero P. Simoncelli",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "ssim",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": "https://ece.uwaterloo.ca/~z70wang/research/ssim/",
    "venue": "IEEE Transactions on Image Processing 13(4), 2004",
    "paperUrl": "https://ieeexplore.ieee.org/document/1284395/",
    "pdfUrl": "https://ece.uwaterloo.ca/~z70wang/publications/ssim.pdf",
    "doi": null,
    "publicationYear": 2004,
    "bibtex": "@article{ssim,\n  title={Image Quality Assessment: From Error Visibility to Structural Similarity},\n  author={Wang, Zhou and Bovik, Alan C. and Sheikh, Hamid R. and Simoncelli, Eero P.},\n  journal={IEEE Transactions on Image Processing},\n  volume={13},\n  number={4},\n  pages={600--612},\n  year={2004}\n}",
    "majorCategory": "评估指标（Metrics）",
    "subcategories": [
      "图像质量指标",
      "评估指标与协议"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "综述明确"
  },
  {
    "id": "ref-00a732613bb51f4fb578",
    "title": "ROUGE: A Package for Automatic Evaluation of Summaries",
    "authors": "Chin-Yew Lin",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "rouge",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "ACL Workshop: Text Summarization Branches Out, 2004",
    "paperUrl": "https://aclanthology.org/W04-1013/",
    "pdfUrl": "https://aclanthology.org/W04-1013.pdf",
    "doi": null,
    "publicationYear": 2004,
    "bibtex": "@inproceedings{rouge,\n  title={{ROUGE}: A Package for Automatic Evaluation of Summaries},\n  author={Lin, Chin-Yew},\n  booktitle={Text Summarization Branches Out},\n  pages={74--81},\n  year={2004}\n}",
    "majorCategory": "评估指标（Metrics）",
    "subcategories": [
      "语言生成指标",
      "评估指标与协议"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "综述明确"
  },
  {
    "id": "ref-58789fde10c5f80c4c89",
    "title": "Real-time humanoid motion generation through ZMP manipulation based on inverted pendulum control",
    "authors": "Tomomichi Sugihara; Yoshihiko Nakamura; Hirochika Inoue",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "sugihara2002real",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "ICRA 2002",
    "paperUrl": "https://ieeexplore.ieee.org/document/1014740",
    "pdfUrl": null,
    "doi": "https://doi.org/10.1109/ROBOT.2002.1014740",
    "publicationYear": 2002,
    "bibtex": "@inproceedings{sugihara2002real,\n  title={Real-time humanoid motion generation through ZMP manipulation based on inverted pendulum control},\n  author={Sugihara, Tomomichi and Nakamura, Yoshihiko and Inoue, Hirochika},\n  booktitle={ICRA},\n  volume={2},\n  pages={1404--1409},\n  year={2002},\n}",
    "majorCategory": "奠基性工作",
    "subcategories": [
      "理论与规划",
      "动作策略基础"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "综述明确"
  },
  {
    "id": "ref-268dde884b1840a3a100",
    "title": "Bleu: a Method for Automatic Evaluation of Machine Translation",
    "authors": "Kishore Papineni; Salim Roukos; Todd Ward; Wei-Jing Zhu",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "bleu",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "ACL 2002",
    "paperUrl": "https://aclanthology.org/P02-1040/",
    "pdfUrl": "https://aclanthology.org/P02-1040.pdf",
    "doi": "https://doi.org/10.3115/1073083.1073135",
    "publicationYear": 2002,
    "bibtex": "@inproceedings{bleu,\n  title={{BLEU}: A Method for Automatic Evaluation of Machine Translation},\n  author={Papineni, Kishore and Roukos, Salim and Ward, Todd and Zhu, Wei-Jing},\n  booktitle={ACL},\n  pages={311--318},\n  year={2002}\n}",
    "majorCategory": "评估指标（Metrics）",
    "subcategories": [
      "语言生成指标",
      "评估指标与协议"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "综述明确"
  },
  {
    "id": "ref-55da7bc51e67237814d2",
    "title": "Planning and acting in partially observable stochastic domains",
    "authors": "Leslie Pack Kaelbling; Michael L. Littman; Anthony R. Cassandra",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "kaelbling1998planning",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "Artificial Intelligence 1998",
    "paperUrl": "https://www.sciencedirect.com/science/article/pii/S000437029800023X",
    "pdfUrl": null,
    "doi": "https://doi.org/10.1016/S0004-3702(98)00023-X",
    "publicationYear": 1998,
    "bibtex": "@article{kaelbling1998planning,\n  author       = {Leslie Pack Kaelbling and\n                  Michael L. Littman and\n                  Anthony R. Cassandra},\n  title        = {Planning and Acting in Partially Observable Stochastic Domains},\n  journal      = {Artif. Intell.},\n  volume       = {101},\n  number       = {1-2},\n  pages        = {99--134},\n  year         = {1998},\n}",
    "majorCategory": "奠基性工作",
    "subcategories": [
      "理论与规划"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "综述明确"
  },
  {
    "id": "ref-206bb9b995e39760f7d0",
    "title": "Reinforcement Learning: An Introduction",
    "authors": "Richard S. Sutton; Andrew G. Barto",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "sutton",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "MIT Press（图书，第一版）",
    "paperUrl": "https://mitpress.mit.edu/9780262193986/reinforcement-learning/",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": 1998,
    "bibtex": "@book{sutton,\n  title={Reinforcement learning: An introduction},\n  author={Sutton, Richard S and Barto, Andrew G and Barto, Andrew},\n  volume={1},\n  number={1},\n  year={1998},\n  publisher={MIT press Cambridge}\n}",
    "majorCategory": "奠基性工作",
    "subcategories": [
      "理论与规划",
      "经典WM与模型式RL"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "综述明确"
  },
  {
    "id": "ref-36bafee9274250697060",
    "title": "Model predictive control: Theory and practice—A survey",
    "authors": "Carlos E. García; David M. Prett; Manfred Morari",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "mpc",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": null,
    "venue": "Automatica",
    "paperUrl": "https://www.sciencedirect.com/science/article/abs/pii/0005109889900022",
    "pdfUrl": null,
    "doi": "https://doi.org/10.1016/0005-1098(89)90002-2",
    "publicationYear": 1989,
    "bibtex": "@article{mpc,\n  author       = {Carlos E. Garcia and\n                  David M. Prett and\n                  Manfred Morari},\n  title        = {Model predictive control: Theory and practice - {A} survey},\n  journal      = {Autom.},\n  volume       = {25},\n  number       = {3},\n  pages        = {335--348},\n  year         = {1989},\n\n}",
    "majorCategory": "奠基性工作",
    "subcategories": [
      "理论与规划",
      "综述与技术资源"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "综述明确"
  },
  {
    "id": "ref-23e2ef710ce5722e25a2",
    "title": "Advancing AI for the physical world",
    "authors": "Microsoft Research",
    "affiliations": "",
    "contribution": "",
    "abstract": "",
    "submittedDate": null,
    "primaryCategory": "Uncategorized",
    "secondaryCategories": [],
    "bibtexKey": "microsoft_advancing_ai",
    "arxivUrl": null,
    "codeUrls": [],
    "projectUrl": "https://www.microsoft.com/en-us/research/story/advancing-ai-for-the-physical-world/",
    "venue": null,
    "paperUrl": "https://www.microsoft.com/en-us/research/story/advancing-ai-for-the-physical-world/",
    "pdfUrl": null,
    "doi": null,
    "publicationYear": null,
    "bibtex": "@misc{microsoft_advancing_ai,\n    title = {Advancing AI for the physical world},\n    author = {{Microsoft Research}},\n    howpublished = {\\url{https://www.microsoft.com/en-us/research/story/advancing-ai-for-the-physical-world/}},\n}",
    "majorCategory": "VLA",
    "subcategories": [
      "多模态触觉VLA"
    ],
    "architecture": "不适用",
    "predictionParadigm": "不适用",
    "quadrant": "不适用",
    "classificationStatus": "综述明确"
  }
]